froid-loop 0.11.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- froid_loop/__init__.py +11 -0
- froid_loop/__main__.py +12 -0
- froid_loop/adapters/__init__.py +3 -0
- froid_loop/adapters/base.py +254 -0
- froid_loop/adapters/entrypoints.py +63 -0
- froid_loop/adapters/env_fault.py +290 -0
- froid_loop/adapters/generic.py +2013 -0
- froid_loop/adapters/mock.py +49 -0
- froid_loop/adapters/multiplexer.py +914 -0
- froid_loop/adapters/opencode_http.py +1687 -0
- froid_loop/adapters/profile.py +650 -0
- froid_loop/adapters/psmux_backend.py +1428 -0
- froid_loop/adapters/registry.py +322 -0
- froid_loop/adapters/tmux_backend.py +35 -0
- froid_loop/adapters/tmux_base.py +630 -0
- froid_loop/checks.py +187 -0
- froid_loop/cli.py +5041 -0
- froid_loop/data/__init__.py +0 -0
- froid_loop/data/froid_loop_hook.py +228 -0
- froid_loop/data/froid_loop_probe_hook.py +88 -0
- froid_loop/data/plugins/example/plugin.toml +21 -0
- froid_loop/data/plugins/tea/plugin.toml +184 -0
- froid_loop/data/plugins/tea/tea_plugin.py +258 -0
- froid_loop/data/plugins/unity/plugin.toml +140 -0
- froid_loop/data/plugins/unity/unity_assets/FroidLoop.Unity.Editor.asmdef +16 -0
- froid_loop/data/plugins/unity/unity_assets/FroidLoop.Unity.Editor.asmdef.meta +7 -0
- froid_loop/data/plugins/unity/unity_assets/SceneAutoSaveGuard.cs +221 -0
- froid_loop/data/plugins/unity/unity_assets/SceneAutoSaveGuard.cs.meta +11 -0
- froid_loop/data/plugins/unity/unity_assets/_folders/Editor.meta +8 -0
- froid_loop/data/plugins/unity/unity_assets/_folders/FroidLoop.meta +8 -0
- froid_loop/data/plugins/unity/unity_cleanup.py +125 -0
- froid_loop/data/plugins/unity/unity_dialog_probe.py +239 -0
- froid_loop/data/plugins/unity/unity_facts.md +17 -0
- froid_loop/data/plugins/unity/unity_plugin.py +415 -0
- froid_loop/data/plugins/unity/unity_quiesce.py +234 -0
- froid_loop/data/plugins/unity/unity_ready.py +230 -0
- froid_loop/data/plugins/unity/unity_seed_assets.py +298 -0
- froid_loop/data/plugins/unity/unity_setup.py +551 -0
- froid_loop/data/plugins/unity/unity_teardown.py +362 -0
- froid_loop/data/profiles/antigravity.toml +52 -0
- froid_loop/data/profiles/claude.toml +85 -0
- froid_loop/data/profiles/codex.toml +22 -0
- froid_loop/data/profiles/copilot.toml +52 -0
- froid_loop/data/profiles/gemini.toml +26 -0
- froid_loop/data/profiles/opencode.toml +54 -0
- froid_loop/data/settings/core.toml +458 -0
- froid_loop/data/skills/README.md +93 -0
- froid_loop/data/skills/froid-loop-resolve/SKILL.md +288 -0
- froid_loop/data/skills/froid-loop-setup/SKILL.md +161 -0
- froid_loop/data/skills/froid-loop-setup/assets/module-help.csv +3 -0
- froid_loop/data/skills/froid-loop-setup/assets/module.yaml +19 -0
- froid_loop/data/skills/froid-loop-sweep/SKILL.md +100 -0
- froid_loop/data/skills/froid-loop-sweep/automation-mode.md +127 -0
- froid_loop/data/skills/froid-loop-sweep/deferred-work-format.md +302 -0
- froid_loop/data/skills/froid-loop-sweep/migration-mode.md +86 -0
- froid_loop/decisions.py +202 -0
- froid_loop/deferredwork.py +2282 -0
- froid_loop/devcontract.py +892 -0
- froid_loop/diagnostics.py +1104 -0
- froid_loop/documents.py +532 -0
- froid_loop/engine.py +7732 -0
- froid_loop/envvars.py +111 -0
- froid_loop/escalation.py +225 -0
- froid_loop/events.py +266 -0
- froid_loop/fences.py +103 -0
- froid_loop/froidconfig.py +226 -0
- froid_loop/frontmatter.py +526 -0
- froid_loop/gates.py +133 -0
- froid_loop/install.py +2936 -0
- froid_loop/journal.py +178 -0
- froid_loop/machine.py +148 -0
- froid_loop/model.py +898 -0
- froid_loop/operatoractions.py +474 -0
- froid_loop/platform_util.py +1490 -0
- froid_loop/plugins/__init__.py +64 -0
- froid_loop/plugins/bus.py +259 -0
- froid_loop/plugins/context.py +319 -0
- froid_loop/plugins/loader.py +145 -0
- froid_loop/plugins/manifest.py +279 -0
- froid_loop/plugins/model.py +296 -0
- froid_loop/plugins/registry.py +245 -0
- froid_loop/plugins/trust.py +75 -0
- froid_loop/policy.py +1569 -0
- froid_loop/probe.py +1044 -0
- froid_loop/process_host.py +408 -0
- froid_loop/recovery_flow.py +1561 -0
- froid_loop/resolve.py +283 -0
- froid_loop/runs.py +4715 -0
- froid_loop/runsetup.py +1293 -0
- froid_loop/sanitize.py +593 -0
- froid_loop/settings_schema.py +276 -0
- froid_loop/signals.py +160 -0
- froid_loop/sprintstatus.py +609 -0
- froid_loop/statemachine.py +57 -0
- froid_loop/stories.py +615 -0
- froid_loop/stories_engine.py +796 -0
- froid_loop/sweep.py +1892 -0
- froid_loop/tokens.py +196 -0
- froid_loop/tui/__init__.py +11 -0
- froid_loop/tui/app.py +1584 -0
- froid_loop/tui/data.py +840 -0
- froid_loop/tui/launch.py +1003 -0
- froid_loop/tui/screens/__init__.py +1 -0
- froid_loop/tui/screens/dashboard.py +1071 -0
- froid_loop/tui/screens/modals.py +943 -0
- froid_loop/tui/screens/settings_screen.py +477 -0
- froid_loop/tui/settings.py +135 -0
- froid_loop/tui/widgets.py +981 -0
- froid_loop/verify.py +4545 -0
- froid_loop/workspace.py +320 -0
- froid_loop/worktree_flow.py +2301 -0
- froid_loop-0.11.1.dist-info/METADATA +728 -0
- froid_loop-0.11.1.dist-info/RECORD +116 -0
- froid_loop-0.11.1.dist-info/WHEEL +4 -0
- froid_loop-0.11.1.dist-info/entry_points.txt +2 -0
- froid_loop-0.11.1.dist-info/licenses/LICENSE +30 -0
froid_loop/tui/launch.py
ADDED
|
@@ -0,0 +1,1003 @@
|
|
|
1
|
+
"""Detached launching of froid-loop commands for the TUI.
|
|
2
|
+
|
|
3
|
+
The TUI never runs engines in-process: run/sweep/resume are launched in new
|
|
4
|
+
windows of a dedicated control session (froid-loop-ctl on tmux, a per-registry
|
|
5
|
+
name on psmux — see the CTL_SESSION comment below) so they survive TUI exit,
|
|
6
|
+
and the dashboard observes them through run-dir artifacts exactly like runs
|
|
7
|
+
started from a plain shell. Fast read-only commands (validate,
|
|
8
|
+
--dry-run) are captured instead, for display in a modal.
|
|
9
|
+
|
|
10
|
+
No textual imports here — everything drives the multiplexer seam (or a plain
|
|
11
|
+
subprocess for the captured read-only commands) and is unit-testable.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import os
|
|
17
|
+
import re
|
|
18
|
+
import stat
|
|
19
|
+
import subprocess
|
|
20
|
+
import sys
|
|
21
|
+
from enum import StrEnum
|
|
22
|
+
from pathlib import Path
|
|
23
|
+
|
|
24
|
+
from .. import runs
|
|
25
|
+
from ..adapters.multiplexer import (
|
|
26
|
+
MultiplexerError,
|
|
27
|
+
get_multiplexer,
|
|
28
|
+
mux_usable,
|
|
29
|
+
)
|
|
30
|
+
from ..journal import Journal
|
|
31
|
+
from ..platform_util import (
|
|
32
|
+
DIR_FD_ANCHORED_WRITES,
|
|
33
|
+
atomic_write_text,
|
|
34
|
+
atomic_write_text_at,
|
|
35
|
+
open_dir_confined,
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
CTL_SESSION = runs.CTL_SESSION
|
|
39
|
+
# The control-session NAME is the transport's business, resolved per call
|
|
40
|
+
# through `runs.ctl_session_for(project)`: the fixed name on tmux, where one
|
|
41
|
+
# server serves the machine and the session really is machine-wide (scoped by
|
|
42
|
+
# the per-window PROJECT_OPTION tag below), and a per-registry name on psmux,
|
|
43
|
+
# whose duplicate-server mutex is keyed on the session name alone, across
|
|
44
|
+
# every registry in the login session (`Local\` is a per-login-session object
|
|
45
|
+
# namespace) — so a fixed name would let only ONE registry there hold a control
|
|
46
|
+
# session and every other project's launch would fail as a duplicate.
|
|
47
|
+
# The constant survives as the fixed base name (display fallbacks, tmux argv
|
|
48
|
+
# pins); anything that addresses a live session resolves the name instead.
|
|
49
|
+
|
|
50
|
+
# control-session windows are named <kind>-<run_id> (see start_detached)
|
|
51
|
+
_CTL_WINDOW_RE = re.compile(r"^(?:run|sweep|resume|resolve)-(.+)$")
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
class LaunchError(Exception):
|
|
55
|
+
pass
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def mux_available() -> bool:
|
|
59
|
+
# Forced-aware (mux_usable, not raw available()): a pinned backend must look
|
|
60
|
+
# the same to observers (attach, ctl-window lookup, prune) as it does to the
|
|
61
|
+
# launch preflight, or a launched run becomes invisible to the rest of the TUI.
|
|
62
|
+
return mux_usable(get_multiplexer())
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def session_exists(session: str) -> bool:
|
|
66
|
+
return get_multiplexer().has_session(session)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
# Run-dir sidecar naming the ctl-session window start_detached minted last for
|
|
70
|
+
# this run. `<kind>-<run_id>` is not unique across the four kinds, so the window
|
|
71
|
+
# listing alone cannot tell a live resume window from the parked run window it
|
|
72
|
+
# superseded — this file names the one we actually created. A hint, never a
|
|
73
|
+
# target on its own: ctl_window_id re-proves it against the live listing.
|
|
74
|
+
_CTL_WINDOW_FILE = "ctl-window"
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
# Generous ceiling on the hint: the value is a window id (`@7`, or a
|
|
78
|
+
# session-qualified `froid-loop-ctl:@7`), and anything longer is already not one.
|
|
79
|
+
_MAX_RECORD_BYTES = 256
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _read_ctl_window(project: Path, run_id: str) -> str | None:
|
|
83
|
+
"""The window id recorded by the run's last launch, or None when there is
|
|
84
|
+
none / it cannot be read. Never raises, and that includes decoding: a torn
|
|
85
|
+
record can raise UnicodeDecodeError, a ValueError rather than an OSError,
|
|
86
|
+
which action_attach (no covering except at all) and _stop_run_worker (whose
|
|
87
|
+
except does not include it) would let escape. An unreadable hint is not an
|
|
88
|
+
error — it just leaves the caller with the name scan.
|
|
89
|
+
|
|
90
|
+
The file is the only channel on purpose: `froid-loop attach` resolves the same
|
|
91
|
+
run from its own process, and one resolve feeding every consumer is the
|
|
92
|
+
property ctl_window_id sells. A per-process memo of what this process last
|
|
93
|
+
minted would answer a different window than the CLI does.
|
|
94
|
+
|
|
95
|
+
Deliberately not `read_text`. The record sits under the project root every
|
|
96
|
+
coding session can write, and this read runs on Textual's event loop
|
|
97
|
+
(`action_attach` calls it directly), so the *shape* of what is at the path
|
|
98
|
+
has to be established before any bytes are consumed:
|
|
99
|
+
|
|
100
|
+
* `O_NONBLOCK` + an `S_ISREG` check on the opened descriptor. Opening a FIFO
|
|
101
|
+
for reading otherwise blocks until someone writes — indefinitely, freezing
|
|
102
|
+
the dashboard on a keypress.
|
|
103
|
+
* `O_NOFOLLOW`, so the name is read rather than wherever it points.
|
|
104
|
+
* At most `_MAX_RECORD_BYTES`. A record pointed at an endless source reads
|
|
105
|
+
forever otherwise, and it raises `MemoryError` rather than the OSError
|
|
106
|
+
this promises never to leak — `Exception` would catch that but also mask
|
|
107
|
+
real bugs, where a cap removes the condition instead of absorbing it.
|
|
108
|
+
|
|
109
|
+
The check is on the descriptor, not the path, so it cannot be raced: fstat
|
|
110
|
+
describes the object actually opened. The POSIX-only flags degrade to 0 on
|
|
111
|
+
win32, which has neither FIFOs at these paths nor O_NOFOLLOW; the size cap
|
|
112
|
+
and the regular-file check carry there on their own."""
|
|
113
|
+
record = runs.run_dir_for(project, run_id) / _CTL_WINDOW_FILE
|
|
114
|
+
flags = os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0) | getattr(os, "O_NONBLOCK", 0)
|
|
115
|
+
flags |= getattr(os, "O_BINARY", 0) # win32: no CRLF translation on the raw fd
|
|
116
|
+
try:
|
|
117
|
+
fd = os.open(record, flags)
|
|
118
|
+
except OSError:
|
|
119
|
+
return None
|
|
120
|
+
try:
|
|
121
|
+
if not stat.S_ISREG(os.fstat(fd).st_mode):
|
|
122
|
+
return None
|
|
123
|
+
data = os.read(fd, _MAX_RECORD_BYTES)
|
|
124
|
+
except OSError:
|
|
125
|
+
return None
|
|
126
|
+
finally:
|
|
127
|
+
os.close(fd)
|
|
128
|
+
try:
|
|
129
|
+
return data.decode("utf-8").strip() or None
|
|
130
|
+
except UnicodeDecodeError:
|
|
131
|
+
return None
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def _forget_ctl_window(project: Path, run_id: str) -> None:
|
|
135
|
+
"""Drop the record. A launch that cannot name the window it just minted must
|
|
136
|
+
not leave the *previous* launch's id authoritative — that id now names a
|
|
137
|
+
superseded window, and the honest answer is no record at all, which puts the
|
|
138
|
+
lookup back on the name scan.
|
|
139
|
+
|
|
140
|
+
Ceiling: when the removal fails, or is declined because the path cannot be
|
|
141
|
+
vouched for, the superseded id survives on disk. It still has to pass
|
|
142
|
+
ctl_window_id's re-prove, so the worst it can answer is a live window
|
|
143
|
+
carrying this run's name — the pre-fix by-name result, never a wilder target.
|
|
144
|
+
Retaining a stale hint is strictly the cheaper failure here, which is why
|
|
145
|
+
this declines rather than deleting on a path it cannot stand behind.
|
|
146
|
+
|
|
147
|
+
Anchored exactly like the write in _record_ctl_window, and for a sharper
|
|
148
|
+
reason: a delete needs no race at all. `unlink` does not follow a link at the
|
|
149
|
+
*final* component, but the ancestors resolve normally, so a run dir standing
|
|
150
|
+
as a link to an external directory makes `run_dir / ctl-window` name a file
|
|
151
|
+
over there — another project's live record — and this deletes it. The write
|
|
152
|
+
path's escape needed the attacker to win a window between check and write;
|
|
153
|
+
a planted link just sits there until the next launch fails to capture an id.
|
|
154
|
+
So the descriptor from `open_dir_confined` is what the unlink is relative to,
|
|
155
|
+
and no path is named. win32 keeps the check-then-delete fallback on the same
|
|
156
|
+
terms as the write — see `_run_dir_is_confined` for that residual.
|
|
157
|
+
|
|
158
|
+
A plain unlink, not retrying_unlink: launches run on the Textual event
|
|
159
|
+
loop, and dropping a best-effort hint is not worth ~5s of blocked win32
|
|
160
|
+
backoff — the ceiling above already covers the miss."""
|
|
161
|
+
run_dir = runs.run_dir_for(project, run_id)
|
|
162
|
+
try:
|
|
163
|
+
if DIR_FD_ANCHORED_WRITES:
|
|
164
|
+
dir_fd = open_dir_confined(project, run_dir)
|
|
165
|
+
if dir_fd is None:
|
|
166
|
+
return # a component we cannot vouch for — see the ceiling
|
|
167
|
+
try:
|
|
168
|
+
os.unlink(_CTL_WINDOW_FILE, dir_fd=dir_fd)
|
|
169
|
+
except FileNotFoundError:
|
|
170
|
+
pass # already gone: missing_ok, by hand
|
|
171
|
+
finally:
|
|
172
|
+
os.close(dir_fd)
|
|
173
|
+
else:
|
|
174
|
+
if not _run_dir_is_confined(project, run_dir):
|
|
175
|
+
return # see the ceiling
|
|
176
|
+
(run_dir / _CTL_WINDOW_FILE).unlink(missing_ok=True)
|
|
177
|
+
except OSError:
|
|
178
|
+
pass # a removal we cannot force — see the ceiling
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def _is_link_of_any_kind(path: Path) -> bool:
|
|
182
|
+
"""Whether `path` is a link that redirects traversal — symlink or, on win32,
|
|
183
|
+
a junction. Raises `OSError` for a component that cannot be probed, which
|
|
184
|
+
the caller turns into a refusal.
|
|
185
|
+
|
|
186
|
+
`is_symlink()` alone is not enough, and the gap is win32-shaped. It answers
|
|
187
|
+
for the symlink reparse tag only and returns **False** for a directory
|
|
188
|
+
junction, which redirects traversal identically. A junction is also the
|
|
189
|
+
*easier* plant of the two: `mklink /J` needs neither elevation nor Developer
|
|
190
|
+
Mode, while a symlink needs one of them. So the check this backs would have
|
|
191
|
+
been blind on win32 to the cheaper version of the very attack it exists for.
|
|
192
|
+
|
|
193
|
+
Detected by the reparse-point attribute rather than `os.path.isjunction`,
|
|
194
|
+
which only exists from 3.12 — this project supports 3.11, and that leg is
|
|
195
|
+
one CI runs on win32. One `lstat`, no version branch: `st_file_attributes`
|
|
196
|
+
is win32-only, so the bit test degrades to False on POSIX, where `S_ISLNK`
|
|
197
|
+
is already the whole answer.
|
|
198
|
+
|
|
199
|
+
Any reparse point counts, not just the junction tag. Other kinds (cloud
|
|
200
|
+
placeholders, app-exec links) have no business being a run dir, and the
|
|
201
|
+
failure this produces is a refusal to write a best-effort hint — the lookup
|
|
202
|
+
degrades to the name scan. Over-refusing is the cheap direction here."""
|
|
203
|
+
info = os.lstat(path)
|
|
204
|
+
if stat.S_ISLNK(info.st_mode):
|
|
205
|
+
return True
|
|
206
|
+
attributes = getattr(info, "st_file_attributes", 0) # win32-only field
|
|
207
|
+
return bool(attributes & stat.FILE_ATTRIBUTE_REPARSE_POINT)
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def _run_dir_is_confined(project: Path, run_dir: Path) -> bool:
|
|
211
|
+
"""Whether `run_dir` is reached from `project` without traversing a link.
|
|
212
|
+
|
|
213
|
+
`follow_symlinks=False` refuses a link at the *final* component only, which
|
|
214
|
+
leaves the ancestors: a session that replaces `.froid-loop/runs/<run_id>`
|
|
215
|
+
with a link to an external directory holding a `state.json` passes
|
|
216
|
+
`runs.is_run` — it follows the link — and then `mkstemp`/`os.replace` land
|
|
217
|
+
the record inside the linked-to directory. The escape is narrower than the
|
|
218
|
+
final-component one (the name written is always `ctl-window`, so the reach
|
|
219
|
+
is another project's record rather than any file), but it is the same shape.
|
|
220
|
+
|
|
221
|
+
Every component below `project` is checked, and `project` itself is not: the
|
|
222
|
+
operator chooses where the project lives and may well keep it behind a link,
|
|
223
|
+
while everything under it is session-writable. `lstat`-based throughout, so
|
|
224
|
+
the check never resolves through what it is testing for.
|
|
225
|
+
|
|
226
|
+
Each component goes through `_is_link_of_any_kind`, not `is_symlink()` —
|
|
227
|
+
on win32 the latter is blind to a junction, which redirects the same way and
|
|
228
|
+
is the easier of the two to plant. That also fixes a quieter gap: `Path`'s
|
|
229
|
+
predicates swallow the `OSError` from a component that cannot be probed and
|
|
230
|
+
answer False, so an unreadable ancestor used to be walked *past* as "not a
|
|
231
|
+
link" — the opposite of the sentence below. Raising from the probe is what
|
|
232
|
+
makes that sentence true.
|
|
233
|
+
|
|
234
|
+
A check, not a race-free open: the portable answer would be to walk the
|
|
235
|
+
components with `dir_fd`, which POSIX has and win32 does not, and this
|
|
236
|
+
record is atomic precisely for the win32 leg. So the standing redirect —
|
|
237
|
+
plant a link, wait for a launch — is what this removes; a session that
|
|
238
|
+
re-plants inside the window between check and write still wins. That
|
|
239
|
+
residual is bounded by the two facts above: same uid as the writer, and a
|
|
240
|
+
fixed filename carrying a window id."""
|
|
241
|
+
try:
|
|
242
|
+
if not run_dir.is_relative_to(project):
|
|
243
|
+
return False
|
|
244
|
+
cursor = run_dir
|
|
245
|
+
while cursor != project:
|
|
246
|
+
if _is_link_of_any_kind(cursor):
|
|
247
|
+
return False
|
|
248
|
+
cursor = cursor.parent
|
|
249
|
+
except OSError:
|
|
250
|
+
return False # a component we cannot probe is one we cannot vouch for
|
|
251
|
+
return True
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
def _record_ctl_window(project: Path, run_id: str, win_id: str) -> None:
|
|
255
|
+
"""Record the window a launch just minted, so ctl_window_id can prefer it
|
|
256
|
+
over an older window sharing the run id.
|
|
257
|
+
|
|
258
|
+
Best-effort on purpose. The window is already running by the time this
|
|
259
|
+
writes, so a failed write must not fail the launch — the lookup degrades to
|
|
260
|
+
the name scan, i.e. to the behaviour before this record existed. A failure
|
|
261
|
+
forgets the previous record rather than leaving it: degrading to the scan is
|
|
262
|
+
the intended fallback, answering a superseded window is not.
|
|
263
|
+
|
|
264
|
+
Skipped when there is no run yet: a fresh `run`/`sweep` mints the only
|
|
265
|
+
window carrying its run id (nothing to disambiguate), and the run dir is
|
|
266
|
+
created by the detached child — this record deliberately never mkdirs one,
|
|
267
|
+
and must not be written into a run-dir-shaped directory (pruned, partial)
|
|
268
|
+
that runs.is_run reports as not a run.
|
|
269
|
+
|
|
270
|
+
That skip forgets too, and the "nothing to disambiguate" clause above is
|
|
271
|
+
exactly why it must. The clause holds for the case it was written for —
|
|
272
|
+
`new_run_id` mints a fresh id, so no other window carries it — but it does
|
|
273
|
+
not hold for every way of reaching this branch. resume/resolve read state,
|
|
274
|
+
raise a confirm modal, and launch from the callback, so anything that
|
|
275
|
+
removes `state.json` inside that human-length window arrives here with a
|
|
276
|
+
predecessor window live and a previous launch's record still on disk. That
|
|
277
|
+
record names the window this launch just superseded, and ctl_window_id
|
|
278
|
+
prefers any record that still resolves — so `a` and `x` would answer the
|
|
279
|
+
parked predecessor while the orchestrator just minted keeps running, which
|
|
280
|
+
is #482's symptom reintroduced by the record meant to fix it. Dropping it
|
|
281
|
+
puts the lookup back on the name scan, which is this file's stated
|
|
282
|
+
preference throughout: degrading to the scan is the intended fallback,
|
|
283
|
+
answering a superseded window is not.
|
|
284
|
+
|
|
285
|
+
Atomic, not a bare write_text: the record is read cross-process (`froid-loop
|
|
286
|
+
attach`), and on win32 an AV/indexer holding the previous record open fails
|
|
287
|
+
a plain overwrite with a transient sharing violation — which would swallow
|
|
288
|
+
into the forget path and quietly degrade the lookup. atomic_replace retries
|
|
289
|
+
exactly that violation, turning most real-world failures into successes.
|
|
290
|
+
|
|
291
|
+
The guard is type-agnostic on purpose, and `OSError` is not wide enough to
|
|
292
|
+
hold it: `atomic_write_text` resolves the path before its own try, and below
|
|
293
|
+
3.13 `Path.resolve` reports a symlink loop as `RuntimeError` — which would
|
|
294
|
+
crash the launch this docstring promises to spare, on the interpreters the
|
|
295
|
+
3.11/3.12 legs run. Same widening, same reason, as the engine's deferred-close
|
|
296
|
+
rollback (`Engine._restore_deferred_closes`). `Exception` and not
|
|
297
|
+
`BaseException`, so a genuine KeyboardInterrupt still gets out.
|
|
298
|
+
|
|
299
|
+
`follow_symlinks=False`, so a symlink at the path is replaced rather than
|
|
300
|
+
written through. Following one is the helper's default contract ("a ledger
|
|
301
|
+
symlinked into the repo keeps being a symlink"), and it is right there — for
|
|
302
|
+
an operator-curated ledger. This sidecar is the opposite: machine-minted,
|
|
303
|
+
per-run, disposable, and living under the project root that every coding
|
|
304
|
+
session can write. Honouring a link here would let a session aim a
|
|
305
|
+
*host-side* write at any path the user can write — reach that the adapters
|
|
306
|
+
confining a session to the workspace otherwise deny it. The payload is only
|
|
307
|
+
a window id, so the primitive is truncation rather than injection, which
|
|
308
|
+
bounds the damage without making it acceptable.
|
|
309
|
+
|
|
310
|
+
Replacing rather than refusing, and no preflight `is_symlink` check: a check
|
|
311
|
+
leaves the window between itself and the write, which a session that
|
|
312
|
+
re-plants the link wins. `os.replace` does not dereference its destination,
|
|
313
|
+
so the link is clobbered whenever it was planted. That also self-heals — the
|
|
314
|
+
record ends up a plain file again — where a refusal would leave the planted
|
|
315
|
+
link in place for the next launch to trip over.
|
|
316
|
+
|
|
317
|
+
Anchored at a directory descriptor where the platform has one. The final
|
|
318
|
+
component is covered by `follow_symlinks=False` above, but the *ancestors*
|
|
319
|
+
are not, and a path check over them (`_run_dir_is_confined`) is answered
|
|
320
|
+
about a path — stale the moment it returns, so a session that re-plants
|
|
321
|
+
`.froid-loop/runs/<run_id>` between check and write still redirects the
|
|
322
|
+
record out of the workspace. `open_dir_confined` walks those components
|
|
323
|
+
`O_NOFOLLOW` and hands back the descriptor for the directory it reached, and
|
|
324
|
+
`atomic_write_text_at` then never names a path again — so a later swap
|
|
325
|
+
renames something this no longer consults, and there is no window to win.
|
|
326
|
+
|
|
327
|
+
win32 keeps the check-then-write path: it has no `*at()` family to anchor
|
|
328
|
+
against (its CPython config defines neither HAVE_RENAMEAT nor HAVE_OPENAT),
|
|
329
|
+
so the descriptor cannot be opened there at all. The residual is documented
|
|
330
|
+
on `_run_dir_is_confined` and bounded by the two facts it names — same uid
|
|
331
|
+
as the writer, and a fixed filename carrying a window id.
|
|
332
|
+
"""
|
|
333
|
+
run_dir = runs.run_dir_for(project, run_id)
|
|
334
|
+
if not runs.is_run(run_dir):
|
|
335
|
+
_forget_ctl_window(project, run_id)
|
|
336
|
+
return
|
|
337
|
+
try:
|
|
338
|
+
if DIR_FD_ANCHORED_WRITES:
|
|
339
|
+
dir_fd = open_dir_confined(project, run_dir)
|
|
340
|
+
if dir_fd is None:
|
|
341
|
+
return # unconfined, or a component we cannot vouch for
|
|
342
|
+
try:
|
|
343
|
+
atomic_write_text_at(dir_fd, _CTL_WINDOW_FILE, win_id)
|
|
344
|
+
finally:
|
|
345
|
+
os.close(dir_fd)
|
|
346
|
+
else:
|
|
347
|
+
if not _run_dir_is_confined(project, run_dir):
|
|
348
|
+
return
|
|
349
|
+
atomic_write_text(run_dir / _CTL_WINDOW_FILE, win_id, follow_symlinks=False)
|
|
350
|
+
except Exception:
|
|
351
|
+
_forget_ctl_window(project, run_id)
|
|
352
|
+
|
|
353
|
+
|
|
354
|
+
def ctl_window_id(project: Path, run_id: str) -> str | None:
|
|
355
|
+
"""Stable window id (bare `@N` on tmux, session-qualified on psmux) of the
|
|
356
|
+
control-session window hosting this run's orchestrator process
|
|
357
|
+
(start_detached names windows <kind>-<run_id>), or None when the run was
|
|
358
|
+
not launched from the TUI or the session is gone.
|
|
359
|
+
|
|
360
|
+
An id, not a name, because every consumer replays the value as a
|
|
361
|
+
select/kill/option target: one resolve feeds all of them, so a rename or a
|
|
362
|
+
window minted between two verbs cannot send them to different windows, and
|
|
363
|
+
the value survives tmux's automatic-rename.
|
|
364
|
+
|
|
365
|
+
`<kind>-<run_id>` is not unique — a resume launched over a still-parked run
|
|
366
|
+
window shares the run id, and nothing reaps the parked one in between — so
|
|
367
|
+
the name scan alone answers whichever match the listing emits first (tmux
|
|
368
|
+
orders by window *index*, and it gives a new window the lowest free index,
|
|
369
|
+
so a superseded window usually but not always sorts ahead of the live one).
|
|
370
|
+
The id the run's last launch minted is recorded in the run dir and wins
|
|
371
|
+
whenever the listing still shows it under this run id. A record that is gone
|
|
372
|
+
(killed, pruned) or now carries another run's name is ignored rather than
|
|
373
|
+
replayed: a target that no longer resolves is the dangerous kind of stale —
|
|
374
|
+
on psmux an unresolvable `-t` lands on the *active* window (psmux/psmux#545;
|
|
375
|
+
tmux merely errors, which the best-effort consumers turn into a silent
|
|
376
|
+
no-op). With no record at all the answer is the first match, exactly as
|
|
377
|
+
before.
|
|
378
|
+
|
|
379
|
+
Scoped to `project` by the PROJECT_OPTION tag, on the same rule as
|
|
380
|
+
_ctl_window_candidates: the control session is shared across projects, and a
|
|
381
|
+
run id is only unique within one (`--run-id` is caller-supplied), so a
|
|
382
|
+
same-id window belonging to another project would otherwise be a legal match
|
|
383
|
+
here — for `x` that means killing a *live* orchestrator next door. An
|
|
384
|
+
untagged window is admitted when this project has the run dir, which keeps a
|
|
385
|
+
window whose (best-effort) tag write failed reachable by its own project
|
|
386
|
+
rather than by nobody.
|
|
387
|
+
|
|
388
|
+
Untagged is a *fallback*, not a peer: an untagged window proves nothing
|
|
389
|
+
about who owns it, so it is consulted only when nothing carries this
|
|
390
|
+
project's tag. Merged into one listing-ordered list they would compete on
|
|
391
|
+
index, and a neighbouring project's untagged window listed first would beat
|
|
392
|
+
this project's correctly tagged one — for `x`, killing next door's
|
|
393
|
+
orchestrator. That case is not hypothetical: the record cannot break the tie
|
|
394
|
+
for a fresh `run`, where recording is deliberately skipped."""
|
|
395
|
+
if not mux_available():
|
|
396
|
+
return None
|
|
397
|
+
mine = runs.accepted_tags(project)
|
|
398
|
+
local = runs.is_run(runs.run_dir_for(project, run_id))
|
|
399
|
+
tagged: list[str] = []
|
|
400
|
+
untagged: list[str] = []
|
|
401
|
+
rows = get_multiplexer().list_windows(
|
|
402
|
+
ctl_session(project), ["window_id", "window_name", runs.PROJECT_OPTION]
|
|
403
|
+
)
|
|
404
|
+
for win_id, name, tag in rows:
|
|
405
|
+
# win_id can be "": psmux's qualifier passes a falsy id through. An
|
|
406
|
+
# empty id must never become a target — an empty `-t` resolves against
|
|
407
|
+
# the *current* window. (The base's short-row padding CAN produce an
|
|
408
|
+
# empty *tag* — it fills trailing fields — which is exactly the untagged
|
|
409
|
+
# case below; window_id stays field 0 of 3.)
|
|
410
|
+
if not win_id:
|
|
411
|
+
continue
|
|
412
|
+
# The whole run id, not a suffix of the name: RUN_ID_RE admits `-`, so
|
|
413
|
+
# `--run-id other-RID` mints `run-other-RID`, which ends with `-RID` and
|
|
414
|
+
# would answer a lookup for `RID` — and sorts ahead of it, so `x` kills
|
|
415
|
+
# the neighbour's LIVE orchestrator. Parsed with the same regex
|
|
416
|
+
# _ctl_window_candidates uses, which also confines a match to the four
|
|
417
|
+
# kinds start_detached mints rather than any name ending this way.
|
|
418
|
+
m = _CTL_WINDOW_RE.match(name)
|
|
419
|
+
if m is None or m.group(1) != run_id:
|
|
420
|
+
continue
|
|
421
|
+
# Set membership, with no "the tag looks unsafe here" escape: the digest
|
|
422
|
+
# arrives as it was written, so a nonempty tag outside the accepted set
|
|
423
|
+
# belongs to another project and must not be a candidate — `x` resolves
|
|
424
|
+
# through here, and admitting a foreign row lets a stop cross a project
|
|
425
|
+
# boundary. The set is what keeps a window tagged by an earlier release
|
|
426
|
+
# reachable: the control session is long-lived and survives the upgrade
|
|
427
|
+
# that changes the tag's spelling, so comparing against the current
|
|
428
|
+
# digest alone would strand this project's own orchestrator — prunable
|
|
429
|
+
# by _ctl_window_candidates, which accepts the legacy tag, yet
|
|
430
|
+
# unreachable by `a` and `x`, which resolve through here.
|
|
431
|
+
if tag in mine:
|
|
432
|
+
tagged.append(win_id)
|
|
433
|
+
elif not tag and local:
|
|
434
|
+
# untagged, and this project holds the run dir — ownership is
|
|
435
|
+
# plausible but unproven, so it only counts if nothing is tagged
|
|
436
|
+
untagged.append(win_id)
|
|
437
|
+
matches = tagged or untagged
|
|
438
|
+
if not matches:
|
|
439
|
+
return None
|
|
440
|
+
# Membership in `matches`, not mere presence in the listing: it re-proves the
|
|
441
|
+
# name and the project too, so neither a backend that reuses a freed window
|
|
442
|
+
# id nor a record naming a neighbouring project's window can be replayed.
|
|
443
|
+
recorded = _read_ctl_window(project, run_id)
|
|
444
|
+
return recorded if recorded in matches else matches[0]
|
|
445
|
+
|
|
446
|
+
|
|
447
|
+
def ctl_window_recorded(project: Path, run_id: str, win_id: str) -> bool:
|
|
448
|
+
"""Whether `ctl_window_id` now answers `win_id` for this run — i.e. whether
|
|
449
|
+
the launch's disambiguation actually took.
|
|
450
|
+
|
|
451
|
+
False means the launch itself succeeded but the lookup is back on the
|
|
452
|
+
ambiguous first-match scan, which is exactly #482's symptom and so is
|
|
453
|
+
operator-visible: every launcher that mints a second window under a run id
|
|
454
|
+
should report it rather than let an unqualified success toast imply the
|
|
455
|
+
targeting is sound. Split out of resume_detached's return so the resolve
|
|
456
|
+
path can warn while still keeping the captured id it attaches with.
|
|
457
|
+
|
|
458
|
+
Asks `ctl_window_id` rather than comparing the record to `win_id`, because
|
|
459
|
+
a round-tripped record is not the same claim. A backend whose
|
|
460
|
+
`new_parked_window` id is shaped differently from its `list_windows`
|
|
461
|
+
`window_id` column — a divergence the seam explicitly tolerates — writes and
|
|
462
|
+
reads the record back intact while `ctl_window_id` rejects it against the
|
|
463
|
+
listing and falls through to the first match. File equality would report
|
|
464
|
+
that as sound; it is the precise case the warning exists for.
|
|
465
|
+
|
|
466
|
+
An unanswerable listing counts as not recorded. The probe is observation, so
|
|
467
|
+
it degrades rather than raising into the launchers (neither has a handler
|
|
468
|
+
for it), and "could not confirm" is closer to the warning's own hedge —
|
|
469
|
+
attach/stop *may* target an older window — than silence would be."""
|
|
470
|
+
try:
|
|
471
|
+
return ctl_window_id(project, run_id) == win_id
|
|
472
|
+
except MultiplexerError:
|
|
473
|
+
return False
|
|
474
|
+
|
|
475
|
+
|
|
476
|
+
def ctl_target(project: Path) -> str:
|
|
477
|
+
"""Seam-canonical target token for this project's control session; see
|
|
478
|
+
:meth:`TerminalMultiplexer.target`. Windows are targeted by stable id
|
|
479
|
+
(ctl_window_id), never by name through this token."""
|
|
480
|
+
return get_multiplexer().target(ctl_session(project))
|
|
481
|
+
|
|
482
|
+
|
|
483
|
+
def select_ctl_window_id(window_id: str) -> None:
|
|
484
|
+
"""Make the window with this id (from start_detached/ctl_window_id) the
|
|
485
|
+
control session's current window, so a plain attach to the session lands
|
|
486
|
+
on it (attach-session itself takes no window)."""
|
|
487
|
+
get_multiplexer().select_window(window_id)
|
|
488
|
+
|
|
489
|
+
|
|
490
|
+
# Per-window tmux user option recording what an interactive attach should do
|
|
491
|
+
# with the client once the window's command exits (consumed by the multiplexer's
|
|
492
|
+
# parked-window return trailer; see start_detached and the tmux backend). Set by
|
|
493
|
+
# set_return_pane at attach time. Value is either a backend-composed pane
|
|
494
|
+
# target — replayed opaquely, so each backend records the form its own
|
|
495
|
+
# switch-client resolves: a bare pane id (%N) on tmux, =session:%N on psmux,
|
|
496
|
+
# whose one-server-per-session model cannot resolve a bare id from the control
|
|
497
|
+
# session (psmux/psmux#483) — used when the TUI runs inside the multiplexer and
|
|
498
|
+
# switched its own client over; or RETURN_DETACH, used when the TUI runs
|
|
499
|
+
# outside and a throwaway client was attached that must detach so the
|
|
500
|
+
# suspended TUI resumes.
|
|
501
|
+
RETURN_OPTION = "@froid_return_pane"
|
|
502
|
+
RETURN_DETACH = "detach" # pane targets are %N / =sess:%N, never "detach"
|
|
503
|
+
|
|
504
|
+
|
|
505
|
+
def current_return_target() -> str | None:
|
|
506
|
+
"""Backend-composed target of the pane this process runs in — the place an
|
|
507
|
+
attach should return the client to — or None when not inside the
|
|
508
|
+
multiplexer / it is unavailable. The value is opaque to callers: record it
|
|
509
|
+
with set_return_pane, replay it via switch_client / the parked trailer.
|
|
510
|
+
See TerminalMultiplexer.current_return_target for the composition
|
|
511
|
+
contract."""
|
|
512
|
+
return get_multiplexer().current_return_target()
|
|
513
|
+
|
|
514
|
+
|
|
515
|
+
def set_return_pane(window_target: str, target: str) -> None:
|
|
516
|
+
"""Record `target` (a current_return_target value or RETURN_DETACH) as the
|
|
517
|
+
return move on a control-session window, so its trailing shell sends the
|
|
518
|
+
client back there when the window's command exits. `window_target` is any
|
|
519
|
+
window spec the backend accepts; callers pass the id from
|
|
520
|
+
start_detached/ctl_window_id so the write lands on the window the caller
|
|
521
|
+
already resolved, not on whatever a fresh by-name lookup answers."""
|
|
522
|
+
get_multiplexer().set_window_option(window_target, RETURN_OPTION, target)
|
|
523
|
+
|
|
524
|
+
|
|
525
|
+
def current_session() -> str | None:
|
|
526
|
+
"""Name of the tmux session this process is running inside, or None when
|
|
527
|
+
not in tmux / tmux is unavailable."""
|
|
528
|
+
return get_multiplexer().current_session()
|
|
529
|
+
|
|
530
|
+
|
|
531
|
+
def in_ctl_session() -> bool:
|
|
532
|
+
"""True when we are running inside a control-session window (i.e. launched
|
|
533
|
+
detached by the TUI), as opposed to a user's own shell. Backend-honest:
|
|
534
|
+
current_session() is None whenever this process is not inside the selected
|
|
535
|
+
multiplexer, so no direct TMUX/HERDR_* env sniffing happens here. The
|
|
536
|
+
shape predicate rather than one project's resolved name: the question its
|
|
537
|
+
callers ask is "am I in A control session", and on a namespacing transport
|
|
538
|
+
the name carries a registry suffix (runs.ctl_session_for)."""
|
|
539
|
+
session = current_session()
|
|
540
|
+
return session is not None and runs.is_ctl_session_name(session)
|
|
541
|
+
|
|
542
|
+
|
|
543
|
+
def detach_client() -> bool:
|
|
544
|
+
"""Detach the tmux client viewing the current session, handing the terminal
|
|
545
|
+
back to the user. Processes in the session keep running. Returns True iff a
|
|
546
|
+
client was actually detached — False both when the transport failed and when
|
|
547
|
+
there was nothing attached (see TerminalMultiplexer.detach_client for how
|
|
548
|
+
each backend establishes that)."""
|
|
549
|
+
return get_multiplexer().detach_client()
|
|
550
|
+
|
|
551
|
+
|
|
552
|
+
class ReturnOutcome(StrEnum):
|
|
553
|
+
"""What return_attached_client managed to do — and, for a caller that goes
|
|
554
|
+
unattended on the strength of it, whether a human can still answer here.
|
|
555
|
+
|
|
556
|
+
A plain boolean cannot carry that: "the hand-back succeeded" and "there is
|
|
557
|
+
still someone at this terminal" are independent, and the ways of failing
|
|
558
|
+
point in different directions. A *refused* switch leaves the client sitting
|
|
559
|
+
in this very window; a switch the backend cannot vouch for may already have
|
|
560
|
+
moved it; a failed *detach* reports no verified hand-back. Three claims, and
|
|
561
|
+
they do not license the same response."""
|
|
562
|
+
|
|
563
|
+
RETURNED = "returned"
|
|
564
|
+
#: No hand-back, but a human may still be here: nothing was recorded to
|
|
565
|
+
#: return to (a plain foreground sweep), the backend is unusable, or the
|
|
566
|
+
#: switch failed with the client still in this window. The conservative
|
|
567
|
+
#: answer — a caller must keep talking to the terminal.
|
|
568
|
+
ATTENDED = "attended"
|
|
569
|
+
#: A hand-back was attempted and did not verifiably happen: the detach found
|
|
570
|
+
#: nothing attached, the switch could not be vouched for (a timed-out verb,
|
|
571
|
+
#: an unreadable client count, nothing attached to move), the effect could
|
|
572
|
+
#: not be observed, or the backend has no detach verb at all (herdr). A
|
|
573
|
+
#: caller must not rely on anyone answering a prompt in this window — a
|
|
574
|
+
#: policy for the uncertainty, not a proof that the window is empty (see
|
|
575
|
+
#: return_attached_client for why it is the safe way to be wrong).
|
|
576
|
+
UNREACHABLE = "unreachable"
|
|
577
|
+
|
|
578
|
+
|
|
579
|
+
def return_attached_client() -> ReturnOutcome:
|
|
580
|
+
"""Hand an attached client back to its origin *now*, mid-process — the
|
|
581
|
+
parked-window return move (see start_detached) executed while the window's
|
|
582
|
+
command keeps running in the background, instead of after it exits.
|
|
583
|
+
|
|
584
|
+
Reads the RETURN_OPTION recorded on the current window by set_return_pane:
|
|
585
|
+
- a pane target (backend-composed: bare %N on tmux, =session:%N on
|
|
586
|
+
psmux): switch that client back there (`-l` fallback if it's gone);
|
|
587
|
+
- RETURN_DETACH: detach the client so a blocking `tmux attach` returns;
|
|
588
|
+
- unset/empty: nobody attached with a return target — do nothing.
|
|
589
|
+
The option is cleared only on RETURNED: a real return must not make the
|
|
590
|
+
parked window's trailer fire a second one, a failed return is left for the
|
|
591
|
+
trailer to retry. That retry is a second chance, not a rescue —
|
|
592
|
+
new_parked_window parks on a blocking read *before* the trailer, so it runs
|
|
593
|
+
only once a human dismisses the park prompt, never in the unattended case.
|
|
594
|
+
|
|
595
|
+
The two failures are not interchangeable, which is why this answers a
|
|
596
|
+
ReturnOutcome and not a bool. A failed switch is positive evidence that the
|
|
597
|
+
client is still in this window, so ATTENDED keeps the caller prompting —
|
|
598
|
+
but only because the seam reserves False for that joint claim. A backend
|
|
599
|
+
that merely cannot vouch for the move (psmux's timed-out verb, an
|
|
600
|
+
unreadable client count, nothing attached to move) answers None, and that
|
|
601
|
+
lands in UNREACHABLE instead. The routing is load-bearing, not tidiness:
|
|
602
|
+
an ATTENDED the client has already walked away from cannot be recovered by
|
|
603
|
+
the surviving return option, because a --repeat cycle prompting into the
|
|
604
|
+
empty window blocks on input() before anyone can reach the trailer. A
|
|
605
|
+
failed detach carries no such evidence in general: on tmux it does
|
|
606
|
+
(`detach-client` fails with "no current client"), but off tmux False also
|
|
607
|
+
covers an effect the backend could not observe and a detach verb it does
|
|
608
|
+
not have at all — herdr, whose False rather than None is exactly what
|
|
609
|
+
detach_client's own widening to a returned bool (#317) buys. That is a
|
|
610
|
+
different widening from switch_client's third state above; detach_client is
|
|
611
|
+
the verb that stays a bool. UNREACHABLE is the policy for all of them, the
|
|
612
|
+
unvouched switch included, because the two ways of being wrong are not
|
|
613
|
+
equally bad: prompting into a
|
|
614
|
+
window no one is viewing blocks a --repeat sweep on input() forever, while
|
|
615
|
+
going unattended in front of a human only defers this cycle's decisions to
|
|
616
|
+
`froid-loop decisions` or the next attended sweep."""
|
|
617
|
+
mux = get_multiplexer()
|
|
618
|
+
if not mux_usable(mux):
|
|
619
|
+
return ReturnOutcome.ATTENDED
|
|
620
|
+
win = mux.current_window_id()
|
|
621
|
+
if win is None:
|
|
622
|
+
return ReturnOutcome.ATTENDED
|
|
623
|
+
ret = mux.show_window_option(win, RETURN_OPTION)
|
|
624
|
+
if not ret:
|
|
625
|
+
return ReturnOutcome.ATTENDED
|
|
626
|
+
if ret == RETURN_DETACH:
|
|
627
|
+
outcome = ReturnOutcome.RETURNED if mux.detach_client() else ReturnOutcome.UNREACHABLE
|
|
628
|
+
else:
|
|
629
|
+
switched = mux.switch_client(ret, last_fallback=True)
|
|
630
|
+
if switched is None:
|
|
631
|
+
outcome = ReturnOutcome.UNREACHABLE
|
|
632
|
+
else:
|
|
633
|
+
outcome = ReturnOutcome.RETURNED if switched else ReturnOutcome.ATTENDED
|
|
634
|
+
if outcome is ReturnOutcome.RETURNED:
|
|
635
|
+
mux.unset_window_option(win, RETURN_OPTION)
|
|
636
|
+
return outcome
|
|
637
|
+
|
|
638
|
+
|
|
639
|
+
def decision_pending(run_dir: Path) -> bool:
|
|
640
|
+
"""True when the run's sweep is currently blocked on an interactive decision
|
|
641
|
+
— its journal's last entry is a decision-pending announcement (the prompter
|
|
642
|
+
blocks on input right after writing it, so any later entry means it moved
|
|
643
|
+
on). Mirrors tui.data.pending_decision; kept here so the CLI can decide an
|
|
644
|
+
attach target without importing the textual-laden data module."""
|
|
645
|
+
entries = Journal(run_dir).entries()
|
|
646
|
+
return bool(entries) and entries[-1].get("kind") == "decision-pending"
|
|
647
|
+
|
|
648
|
+
|
|
649
|
+
def attach_plan(project: Path, run_id: str) -> tuple[list[str], str | None] | None:
|
|
650
|
+
"""Pick where an interactive attach should land for this run and which window
|
|
651
|
+
(if any) to record a return target on. Shared by the CLI `attach` command and
|
|
652
|
+
mirroring the TUI's action_attach logic: prefer the orchestrator's ctl window
|
|
653
|
+
when a sweep is blocked on a decision or no agent session is live, else the
|
|
654
|
+
live agent session. Returns (tmux argv, return_window) or None when there is
|
|
655
|
+
nothing to attach to."""
|
|
656
|
+
session = runs.session_name(run_id)
|
|
657
|
+
win_id = ctl_window_id(project, run_id)
|
|
658
|
+
agent_live = session_exists(session)
|
|
659
|
+
if win_id is not None and (
|
|
660
|
+
decision_pending(runs.run_dir_for(project, run_id)) or not agent_live
|
|
661
|
+
):
|
|
662
|
+
select_ctl_window_id(win_id)
|
|
663
|
+
return runs.attach_target_argv(ctl_target(project)), win_id
|
|
664
|
+
if agent_live:
|
|
665
|
+
return runs.attach_target_argv(runs.session_target(run_id)), None
|
|
666
|
+
return None
|
|
667
|
+
|
|
668
|
+
|
|
669
|
+
def kill_ctl_window(project: Path, run_id: str) -> None:
|
|
670
|
+
"""Kill the control-session window hosting this run's orchestrator process,
|
|
671
|
+
if any. A no-op when the run was not launched from the TUI or tmux is gone."""
|
|
672
|
+
win_id = ctl_window_id(project, run_id)
|
|
673
|
+
if win_id is not None:
|
|
674
|
+
get_multiplexer().kill_window(win_id)
|
|
675
|
+
|
|
676
|
+
|
|
677
|
+
def _ctl_window_candidates(project: Path) -> list[tuple[str, str]]:
|
|
678
|
+
"""(window_id, window_name) for parked control-session run windows whose run
|
|
679
|
+
is no longer live — the kill candidates for a prune.
|
|
680
|
+
|
|
681
|
+
A `<kind>-<run_id>` window parks on a `read` prompt that never closes on its
|
|
682
|
+
own; it is a candidate once its run has finished/stopped/crashed (or its run
|
|
683
|
+
dir is gone). The current window is excluded so a prune triggered from inside
|
|
684
|
+
the ctl session never targets itself; live runs and the session's own shell
|
|
685
|
+
window are excluded too.
|
|
686
|
+
|
|
687
|
+
The control session is shared across projects, so its per-window PROJECT_OPTION
|
|
688
|
+
accepts current and legacy project tags; untagged windows still require a run
|
|
689
|
+
directory under this project (mirrors runs.prunable_sessions).
|
|
690
|
+
"""
|
|
691
|
+
mux = get_multiplexer()
|
|
692
|
+
ctl = runs.ctl_session_for(project, mux)
|
|
693
|
+
if not mux_usable(mux) or not session_exists(ctl):
|
|
694
|
+
return []
|
|
695
|
+
current = mux.current_window_id()
|
|
696
|
+
rows = mux.list_windows(ctl, ["window_id", "window_name", runs.PROJECT_OPTION])
|
|
697
|
+
mine = runs.accepted_tags(project)
|
|
698
|
+
candidates: list[tuple[str, str]] = []
|
|
699
|
+
for win_id, name, tag in rows:
|
|
700
|
+
if not win_id or win_id == current:
|
|
701
|
+
continue
|
|
702
|
+
m = _CTL_WINDOW_RE.match(name)
|
|
703
|
+
if m is None:
|
|
704
|
+
continue # not a run window (e.g. the session's initial shell)
|
|
705
|
+
if not runs.is_parsable_run_id(m.group(1)):
|
|
706
|
+
# A foreign/mangled window name must not steer a run-dir path. The
|
|
707
|
+
# PARSE-side predicate: this window already exists, so the mint's
|
|
708
|
+
# broad ctl reservation would leak every pre-upgrade `run-ctl-*`
|
|
709
|
+
# window out of the sweep instead of closing it.
|
|
710
|
+
continue
|
|
711
|
+
run_dir = runs.run_dir_for(project, m.group(1))
|
|
712
|
+
if tag:
|
|
713
|
+
if tag not in mine:
|
|
714
|
+
continue # another project's window
|
|
715
|
+
elif not runs.is_run(run_dir):
|
|
716
|
+
continue # untagged and no run dir here — ownership unprovable
|
|
717
|
+
# boolean gate on purpose: an 'unknown' engine stays a candidate (unknown
|
|
718
|
+
# never blocks cleanup) with no per-window warning — the session-level
|
|
719
|
+
# unknown warning from prunable_sessions covers the operator surface.
|
|
720
|
+
if runs.engine_alive(run_dir):
|
|
721
|
+
continue
|
|
722
|
+
candidates.append((win_id, name))
|
|
723
|
+
return candidates
|
|
724
|
+
|
|
725
|
+
|
|
726
|
+
def prunable_ctl_windows(project: Path) -> list[str]:
|
|
727
|
+
"""Names of the control-session windows a prune would close (dry-run view)."""
|
|
728
|
+
return [name for _, name in _ctl_window_candidates(project)]
|
|
729
|
+
|
|
730
|
+
|
|
731
|
+
def prune_ctl_windows(project: Path) -> tuple[list[str], list[str], list[str]]:
|
|
732
|
+
"""Close parked control-session windows whose run is no longer live; returns
|
|
733
|
+
(removed, survived, unverifiable) window names (see _ctl_window_candidates).
|
|
734
|
+
A three-list tuple like runs.prune_sessions, but do NOT read the arms across:
|
|
735
|
+
that one partitions BEFORE its kills, so its `killed` is still an attempted
|
|
736
|
+
kill, its `live` is "deliberately not touched" rather than "survived", and its
|
|
737
|
+
`unknown` is a pid question and a SUBSET of `killed`. These three are disjoint
|
|
738
|
+
(by window id — the values are names) and all three are about kill outcome.
|
|
739
|
+
|
|
740
|
+
kill_window is best-effort by contract (a hang, a missing binary, and a
|
|
741
|
+
refused kill are all the same silent no-op), so an attempted kill is not a
|
|
742
|
+
removal and must not be reported as one (#435). The verdict is taken here
|
|
743
|
+
rather than pushed into the seam because this is the caller that both needs
|
|
744
|
+
it and already holds the session: kill_window(target) alone cannot verify
|
|
745
|
+
anything on a backend whose liveness listing is session-scoped, which is all
|
|
746
|
+
of them.
|
|
747
|
+
|
|
748
|
+
ONE listing after the whole fan-out, not a probe per window: the answer is a
|
|
749
|
+
set membership either way, so the verdict costs one extra round trip instead
|
|
750
|
+
of N. A transport fault raises and nothing can be claimed there.
|
|
751
|
+
|
|
752
|
+
Two ceilings, both deliberate:
|
|
753
|
+
|
|
754
|
+
- The membership test pairs list_windows' `window_id` column with
|
|
755
|
+
list_window_ids. The seam states its symmetry rules pairwise and this pair
|
|
756
|
+
is stated because of THIS caller — a backend qualifying one side and not
|
|
757
|
+
the other reads every candidate as removed, which is #435 restored on the
|
|
758
|
+
optimistic side, with no error anywhere.
|
|
759
|
+
- `[]` is read as "the session went with its last window". The seam's `[]`
|
|
760
|
+
is wider than that: BaseTmuxBackend folds EVERY nonzero exit to `[]`, so a
|
|
761
|
+
server that errors while its windows live would report them removed. The
|
|
762
|
+
common cause by far is the session really being gone, and pessimism there
|
|
763
|
+
would invent a phantom survivor on every future sweep. Narrowing the
|
|
764
|
+
sentinel is a change to the engine's liveness probe, not to this function.
|
|
765
|
+
"""
|
|
766
|
+
mux = get_multiplexer()
|
|
767
|
+
candidates = _ctl_window_candidates(project)
|
|
768
|
+
if not candidates:
|
|
769
|
+
return [], [], []
|
|
770
|
+
for win_id, _name in candidates:
|
|
771
|
+
# kill_window is best-effort and reports nothing; a strict-POSIX decode
|
|
772
|
+
# fault of the kill's own capture escapes its swallow tuple (#380) but
|
|
773
|
+
# says exactly as little about the outcome — the command may well have
|
|
774
|
+
# reached the server. More of the same nothing: the fan-out continues
|
|
775
|
+
# and the one post-kill listing hands down the verdict either way. An
|
|
776
|
+
# escape here would surface at the callers as a scan failure, an
|
|
777
|
+
# empty-armed receipt denying kills that just fired.
|
|
778
|
+
try:
|
|
779
|
+
mux.kill_window(win_id)
|
|
780
|
+
except UnicodeError:
|
|
781
|
+
pass
|
|
782
|
+
try:
|
|
783
|
+
live = set(mux.list_window_ids(runs.ctl_session_for(project, mux)))
|
|
784
|
+
except MultiplexerError:
|
|
785
|
+
# The kills may well have landed; nothing here can say so. Claiming the
|
|
786
|
+
# optimistic half is exactly the bug — the next cleanup pass retries.
|
|
787
|
+
return [], [], [name for _win_id, name in candidates]
|
|
788
|
+
removed = [name for win_id, name in candidates if win_id not in live]
|
|
789
|
+
survived = [name for win_id, name in candidates if win_id in live]
|
|
790
|
+
return removed, survived, []
|
|
791
|
+
|
|
792
|
+
|
|
793
|
+
def ctl_session(project: Path) -> str:
|
|
794
|
+
"""The control-session name for this project on the selected transport
|
|
795
|
+
(see `runs.ctl_session_for`). The TUI-facing spelling, so `tui/app.py`
|
|
796
|
+
and the screens name the session an operator would actually attach to."""
|
|
797
|
+
return runs.ctl_session_for(project, get_multiplexer())
|
|
798
|
+
|
|
799
|
+
|
|
800
|
+
def _ensure_ctl_session(project: Path) -> str:
|
|
801
|
+
mux = get_multiplexer()
|
|
802
|
+
name = runs.ctl_session_for(project, mux)
|
|
803
|
+
# has_session is raiser-side (a server-backed backend can fail the probe after
|
|
804
|
+
# the availability pre-gate). Keep it inside the try so a transport failure
|
|
805
|
+
# converts to LaunchError, which the TUI launch/resume/resolve handlers already
|
|
806
|
+
# catch — otherwise the raw MultiplexerError slips past them and crashes the app.
|
|
807
|
+
try:
|
|
808
|
+
if not mux.has_session(name):
|
|
809
|
+
mux.new_session(name, project)
|
|
810
|
+
except MultiplexerError as e:
|
|
811
|
+
raise LaunchError(f"multiplexer ctl-session setup failed: {e}") from e
|
|
812
|
+
return name
|
|
813
|
+
|
|
814
|
+
|
|
815
|
+
def cli_argv(*tail: str) -> list[str]:
|
|
816
|
+
"""`sys.executable -m froid_loop.cli ...` — immune to PATH/venv drift
|
|
817
|
+
inside tmux windows."""
|
|
818
|
+
return [sys.executable, "-m", "froid_loop.cli", *tail]
|
|
819
|
+
|
|
820
|
+
|
|
821
|
+
def start_detached(project: Path, argv_tail: list[str], run_id: str, kind: str) -> str | None:
|
|
822
|
+
"""Run a froid-loop command in a new window of the control session.
|
|
823
|
+
|
|
824
|
+
The window parks after the command exits (keeping the exit status
|
|
825
|
+
inspectable) and then returns an attached client to its origin pane — both
|
|
826
|
+
handled by the multiplexer's parked-window primitive, keyed by the
|
|
827
|
+
RETURN_OPTION recorded on the window by set_return_pane.
|
|
828
|
+
|
|
829
|
+
Returns the new window's stable backend id (bare `@N` on tmux,
|
|
830
|
+
session-qualified on psmux) so callers can target it unambiguously (window
|
|
831
|
+
names collide when several kinds share a run_id). The same id is recorded in
|
|
832
|
+
the run dir so ctl_window_id answers this window rather than an older one
|
|
833
|
+
under the same run id — see _record_ctl_window.
|
|
834
|
+
|
|
835
|
+
Refuses a run id that aliases a control session, FIRST — this is the one
|
|
836
|
+
place every drive path converges on the mutation (the window mint and the
|
|
837
|
+
ctl-window record overwrite): run/sweep launches with freshly validated
|
|
838
|
+
ids, and resume/resolve replaying ids an older release persisted. Gating
|
|
839
|
+
each button separately kept finding the path nobody gated (resolve was
|
|
840
|
+
the fourth); gating the mutation cannot. Ahead of the mux probes so the
|
|
841
|
+
refusal needs no transport to be phrased.
|
|
842
|
+
"""
|
|
843
|
+
if runs.run_id_aliases_control_session(run_id):
|
|
844
|
+
raise LaunchError(
|
|
845
|
+
f"run {run_id}: its agent session name is the control session's own — "
|
|
846
|
+
f"cannot be driven. Recover its work by hand, then `froid-loop delete {run_id}`"
|
|
847
|
+
)
|
|
848
|
+
mux = get_multiplexer()
|
|
849
|
+
if not mux_usable(mux):
|
|
850
|
+
raise LaunchError(
|
|
851
|
+
"multiplexer backend unavailable (binary missing, version unsupported, "
|
|
852
|
+
"or a required helper absent)"
|
|
853
|
+
)
|
|
854
|
+
ctl = _ensure_ctl_session(project)
|
|
855
|
+
try:
|
|
856
|
+
win_id = (
|
|
857
|
+
mux.new_parked_window(
|
|
858
|
+
ctl,
|
|
859
|
+
f"{kind}-{run_id}",
|
|
860
|
+
project,
|
|
861
|
+
cli_argv(*argv_tail),
|
|
862
|
+
RETURN_OPTION,
|
|
863
|
+
)
|
|
864
|
+
or None
|
|
865
|
+
)
|
|
866
|
+
except MultiplexerError as e:
|
|
867
|
+
raise LaunchError(f"multiplexer new-window failed: {e}") from e
|
|
868
|
+
if win_id:
|
|
869
|
+
# Record before tagging: a window minted but unrecorded puts the lookup
|
|
870
|
+
# back on the ambiguous scan, while an *untagged* window already has a
|
|
871
|
+
# documented fallback in _ctl_window_candidates — so even a
|
|
872
|
+
# non-conforming backend raising from the (contractually best-effort)
|
|
873
|
+
# set_window_option must not cost the record.
|
|
874
|
+
_record_ctl_window(project, run_id, win_id)
|
|
875
|
+
# Tag the window with its project so a cleanup in another project never
|
|
876
|
+
# closes it (the ctl session is shared across projects).
|
|
877
|
+
mux.set_window_option(win_id, runs.PROJECT_OPTION, runs.project_tag(project))
|
|
878
|
+
else:
|
|
879
|
+
# No id to record: the backend did not capture one. Whatever the previous
|
|
880
|
+
# launch recorded now names a superseded window, so drop it.
|
|
881
|
+
_forget_ctl_window(project, run_id)
|
|
882
|
+
return win_id
|
|
883
|
+
|
|
884
|
+
|
|
885
|
+
def start_run_detached(
|
|
886
|
+
project: Path,
|
|
887
|
+
run_id: str,
|
|
888
|
+
*,
|
|
889
|
+
spec: str | None = None,
|
|
890
|
+
epic: int | None = None,
|
|
891
|
+
story: str | None = None,
|
|
892
|
+
max_stories: int | None = None,
|
|
893
|
+
) -> None:
|
|
894
|
+
tail = ["run", "--project", str(project), "--run-id", run_id]
|
|
895
|
+
if spec:
|
|
896
|
+
tail += ["--spec", spec] # forces stories mode (folder+id dispatch)
|
|
897
|
+
if epic is not None:
|
|
898
|
+
tail += ["--epic", str(epic)]
|
|
899
|
+
if story:
|
|
900
|
+
tail += ["--story", story]
|
|
901
|
+
if max_stories is not None:
|
|
902
|
+
tail += ["--max-stories", str(max_stories)]
|
|
903
|
+
start_detached(project, tail, run_id, "run")
|
|
904
|
+
|
|
905
|
+
|
|
906
|
+
def start_sweep_detached(
|
|
907
|
+
project: Path,
|
|
908
|
+
run_id: str,
|
|
909
|
+
*,
|
|
910
|
+
no_prompt: bool = False,
|
|
911
|
+
decisions_only: bool = False,
|
|
912
|
+
max_bundles: int | None = None,
|
|
913
|
+
) -> None:
|
|
914
|
+
tail = ["sweep", "--project", str(project), "--run-id", run_id]
|
|
915
|
+
if no_prompt:
|
|
916
|
+
tail.append("--no-prompt")
|
|
917
|
+
if decisions_only:
|
|
918
|
+
tail.append("--decisions-only")
|
|
919
|
+
if max_bundles is not None:
|
|
920
|
+
tail += ["--max-bundles", str(max_bundles)]
|
|
921
|
+
start_detached(project, tail, run_id, "sweep")
|
|
922
|
+
|
|
923
|
+
|
|
924
|
+
def resume_detached(project: Path, run_id: str) -> str | None:
|
|
925
|
+
"""Resume in a ctl-session window; returns the window id, or None when the
|
|
926
|
+
lookup cannot name that window afterwards — the caller should warn, because
|
|
927
|
+
resume is the launch that mints a *second* window under the run id, so this
|
|
928
|
+
is exactly when the ambiguous scan starts answering the superseded one while
|
|
929
|
+
the launch itself succeeded.
|
|
930
|
+
|
|
931
|
+
Two ways to land there, one signal: the backend captured no id, or it did but
|
|
932
|
+
the record did not survive (refused, unwritable, run dir pruned mid-launch).
|
|
933
|
+
Both leave `ctl_window_id` on the scan, so reporting only the first would let
|
|
934
|
+
the rest degrade behind an unqualified success toast.
|
|
935
|
+
|
|
936
|
+
Verified by re-reading rather than by threading the write's outcome up: it
|
|
937
|
+
asks the question the consumers actually ask — will `ctl_window_id` prefer
|
|
938
|
+
this window — of the same file they will read, instead of a proxy for it.
|
|
939
|
+
|
|
940
|
+
Folded into the return here, rather than reported alongside the id as the
|
|
941
|
+
resolve path does, because resume has no immediate use for a window it
|
|
942
|
+
cannot record: it launches and leaves, where resolve attaches to the window
|
|
943
|
+
it just minted and still needs that id to do so."""
|
|
944
|
+
win_id = start_detached(
|
|
945
|
+
project, ["resume", "--project", str(project), run_id], run_id, "resume"
|
|
946
|
+
)
|
|
947
|
+
if win_id and not ctl_window_recorded(project, run_id, win_id):
|
|
948
|
+
return None
|
|
949
|
+
return win_id
|
|
950
|
+
|
|
951
|
+
|
|
952
|
+
def start_resolve_detached(project: Path, run_id: str) -> str | None:
|
|
953
|
+
"""Run `froid-loop resolve <run_id>` in a ctl-session window. The caller
|
|
954
|
+
attaches to it: the resolve agent is interactive, and the post-session
|
|
955
|
+
confirm + resume happen in that same window. Returns the window id so the
|
|
956
|
+
caller attaches to exactly this window, not a stale same-run_id window."""
|
|
957
|
+
return start_detached(
|
|
958
|
+
project, ["resolve", "--project", str(project), run_id], run_id, "resolve"
|
|
959
|
+
)
|
|
960
|
+
|
|
961
|
+
|
|
962
|
+
def run_captured_streams(argv_tail: list[str]) -> tuple[int, str, str]:
|
|
963
|
+
"""Run a fast read-only command (validate, --dry-run) and capture its output
|
|
964
|
+
with the two streams kept **apart**.
|
|
965
|
+
|
|
966
|
+
Separation is the whole point of this seam. Anything parsing stdout as a
|
|
967
|
+
whole document — a ``--json`` command, whose contract in :mod:`froid_loop.machine`
|
|
968
|
+
is that stdout is one JSON object and nothing else — cannot use the merged
|
|
969
|
+
form: :func:`run_captured` appends stderr *after* stdout, so a single
|
|
970
|
+
``DeprecationWarning`` written to the child's stderr by any dependency turns
|
|
971
|
+
``json.loads`` into ``Extra data:``. That failure is environment-dependent —
|
|
972
|
+
it needs the right interpreter, the right installed versions, the right
|
|
973
|
+
warning filters — so it would pass everywhere it was tested and silently
|
|
974
|
+
degrade the JSON path to the text one on a user's machine. A caller that
|
|
975
|
+
genuinely wants one blob merges them itself; a caller that parses must never
|
|
976
|
+
have been handed the option.
|
|
977
|
+
|
|
978
|
+
Decoding is pinned to UTF-8 with ``errors="replace"`` rather than
|
|
979
|
+
``text=True``, which decodes with the *locale* encoding at ``errors="strict"``
|
|
980
|
+
— the #200 family of failure already fixed CLI-side in :mod:`froid_loop.machine`.
|
|
981
|
+
A console in a non-UTF-8 code page must not turn a perfectly good document
|
|
982
|
+
into a ``UnicodeDecodeError`` on the way in.
|
|
983
|
+
"""
|
|
984
|
+
proc = subprocess.run(
|
|
985
|
+
cli_argv(*argv_tail), capture_output=True, encoding="utf-8", errors="replace"
|
|
986
|
+
)
|
|
987
|
+
return proc.returncode, proc.stdout, proc.stderr
|
|
988
|
+
|
|
989
|
+
|
|
990
|
+
def run_captured(argv_tail: list[str]) -> tuple[int, str]:
|
|
991
|
+
"""Run a fast read-only command (validate, --dry-run) and capture its
|
|
992
|
+
combined output for display.
|
|
993
|
+
|
|
994
|
+
For text display only. Anything that parses the output must call
|
|
995
|
+
:func:`run_captured_streams` — see its docstring on why the merge is
|
|
996
|
+
unparseable.
|
|
997
|
+
"""
|
|
998
|
+
rc, out, err = run_captured_streams(argv_tail)
|
|
999
|
+
if err:
|
|
1000
|
+
if out and not out.endswith("\n"):
|
|
1001
|
+
out += "\n"
|
|
1002
|
+
out += err
|
|
1003
|
+
return rc, out
|