froid-loop 0.11.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. froid_loop/__init__.py +11 -0
  2. froid_loop/__main__.py +12 -0
  3. froid_loop/adapters/__init__.py +3 -0
  4. froid_loop/adapters/base.py +254 -0
  5. froid_loop/adapters/entrypoints.py +63 -0
  6. froid_loop/adapters/env_fault.py +290 -0
  7. froid_loop/adapters/generic.py +2013 -0
  8. froid_loop/adapters/mock.py +49 -0
  9. froid_loop/adapters/multiplexer.py +914 -0
  10. froid_loop/adapters/opencode_http.py +1687 -0
  11. froid_loop/adapters/profile.py +650 -0
  12. froid_loop/adapters/psmux_backend.py +1428 -0
  13. froid_loop/adapters/registry.py +322 -0
  14. froid_loop/adapters/tmux_backend.py +35 -0
  15. froid_loop/adapters/tmux_base.py +630 -0
  16. froid_loop/checks.py +187 -0
  17. froid_loop/cli.py +5041 -0
  18. froid_loop/data/__init__.py +0 -0
  19. froid_loop/data/froid_loop_hook.py +228 -0
  20. froid_loop/data/froid_loop_probe_hook.py +88 -0
  21. froid_loop/data/plugins/example/plugin.toml +21 -0
  22. froid_loop/data/plugins/tea/plugin.toml +184 -0
  23. froid_loop/data/plugins/tea/tea_plugin.py +258 -0
  24. froid_loop/data/plugins/unity/plugin.toml +140 -0
  25. froid_loop/data/plugins/unity/unity_assets/FroidLoop.Unity.Editor.asmdef +16 -0
  26. froid_loop/data/plugins/unity/unity_assets/FroidLoop.Unity.Editor.asmdef.meta +7 -0
  27. froid_loop/data/plugins/unity/unity_assets/SceneAutoSaveGuard.cs +221 -0
  28. froid_loop/data/plugins/unity/unity_assets/SceneAutoSaveGuard.cs.meta +11 -0
  29. froid_loop/data/plugins/unity/unity_assets/_folders/Editor.meta +8 -0
  30. froid_loop/data/plugins/unity/unity_assets/_folders/FroidLoop.meta +8 -0
  31. froid_loop/data/plugins/unity/unity_cleanup.py +125 -0
  32. froid_loop/data/plugins/unity/unity_dialog_probe.py +239 -0
  33. froid_loop/data/plugins/unity/unity_facts.md +17 -0
  34. froid_loop/data/plugins/unity/unity_plugin.py +415 -0
  35. froid_loop/data/plugins/unity/unity_quiesce.py +234 -0
  36. froid_loop/data/plugins/unity/unity_ready.py +230 -0
  37. froid_loop/data/plugins/unity/unity_seed_assets.py +298 -0
  38. froid_loop/data/plugins/unity/unity_setup.py +551 -0
  39. froid_loop/data/plugins/unity/unity_teardown.py +362 -0
  40. froid_loop/data/profiles/antigravity.toml +52 -0
  41. froid_loop/data/profiles/claude.toml +85 -0
  42. froid_loop/data/profiles/codex.toml +22 -0
  43. froid_loop/data/profiles/copilot.toml +52 -0
  44. froid_loop/data/profiles/gemini.toml +26 -0
  45. froid_loop/data/profiles/opencode.toml +54 -0
  46. froid_loop/data/settings/core.toml +458 -0
  47. froid_loop/data/skills/README.md +93 -0
  48. froid_loop/data/skills/froid-loop-resolve/SKILL.md +288 -0
  49. froid_loop/data/skills/froid-loop-setup/SKILL.md +161 -0
  50. froid_loop/data/skills/froid-loop-setup/assets/module-help.csv +3 -0
  51. froid_loop/data/skills/froid-loop-setup/assets/module.yaml +19 -0
  52. froid_loop/data/skills/froid-loop-sweep/SKILL.md +100 -0
  53. froid_loop/data/skills/froid-loop-sweep/automation-mode.md +127 -0
  54. froid_loop/data/skills/froid-loop-sweep/deferred-work-format.md +302 -0
  55. froid_loop/data/skills/froid-loop-sweep/migration-mode.md +86 -0
  56. froid_loop/decisions.py +202 -0
  57. froid_loop/deferredwork.py +2282 -0
  58. froid_loop/devcontract.py +892 -0
  59. froid_loop/diagnostics.py +1104 -0
  60. froid_loop/documents.py +532 -0
  61. froid_loop/engine.py +7732 -0
  62. froid_loop/envvars.py +111 -0
  63. froid_loop/escalation.py +225 -0
  64. froid_loop/events.py +266 -0
  65. froid_loop/fences.py +103 -0
  66. froid_loop/froidconfig.py +226 -0
  67. froid_loop/frontmatter.py +526 -0
  68. froid_loop/gates.py +133 -0
  69. froid_loop/install.py +2936 -0
  70. froid_loop/journal.py +178 -0
  71. froid_loop/machine.py +148 -0
  72. froid_loop/model.py +898 -0
  73. froid_loop/operatoractions.py +474 -0
  74. froid_loop/platform_util.py +1490 -0
  75. froid_loop/plugins/__init__.py +64 -0
  76. froid_loop/plugins/bus.py +259 -0
  77. froid_loop/plugins/context.py +319 -0
  78. froid_loop/plugins/loader.py +145 -0
  79. froid_loop/plugins/manifest.py +279 -0
  80. froid_loop/plugins/model.py +296 -0
  81. froid_loop/plugins/registry.py +245 -0
  82. froid_loop/plugins/trust.py +75 -0
  83. froid_loop/policy.py +1569 -0
  84. froid_loop/probe.py +1044 -0
  85. froid_loop/process_host.py +408 -0
  86. froid_loop/recovery_flow.py +1561 -0
  87. froid_loop/resolve.py +283 -0
  88. froid_loop/runs.py +4715 -0
  89. froid_loop/runsetup.py +1293 -0
  90. froid_loop/sanitize.py +593 -0
  91. froid_loop/settings_schema.py +276 -0
  92. froid_loop/signals.py +160 -0
  93. froid_loop/sprintstatus.py +609 -0
  94. froid_loop/statemachine.py +57 -0
  95. froid_loop/stories.py +615 -0
  96. froid_loop/stories_engine.py +796 -0
  97. froid_loop/sweep.py +1892 -0
  98. froid_loop/tokens.py +196 -0
  99. froid_loop/tui/__init__.py +11 -0
  100. froid_loop/tui/app.py +1584 -0
  101. froid_loop/tui/data.py +840 -0
  102. froid_loop/tui/launch.py +1003 -0
  103. froid_loop/tui/screens/__init__.py +1 -0
  104. froid_loop/tui/screens/dashboard.py +1071 -0
  105. froid_loop/tui/screens/modals.py +943 -0
  106. froid_loop/tui/screens/settings_screen.py +477 -0
  107. froid_loop/tui/settings.py +135 -0
  108. froid_loop/tui/widgets.py +981 -0
  109. froid_loop/verify.py +4545 -0
  110. froid_loop/workspace.py +320 -0
  111. froid_loop/worktree_flow.py +2301 -0
  112. froid_loop-0.11.1.dist-info/METADATA +728 -0
  113. froid_loop-0.11.1.dist-info/RECORD +116 -0
  114. froid_loop-0.11.1.dist-info/WHEEL +4 -0
  115. froid_loop-0.11.1.dist-info/entry_points.txt +2 -0
  116. froid_loop-0.11.1.dist-info/licenses/LICENSE +30 -0
@@ -0,0 +1,1003 @@
1
+ """Detached launching of froid-loop commands for the TUI.
2
+
3
+ The TUI never runs engines in-process: run/sweep/resume are launched in new
4
+ windows of a dedicated control session (froid-loop-ctl on tmux, a per-registry
5
+ name on psmux — see the CTL_SESSION comment below) so they survive TUI exit,
6
+ and the dashboard observes them through run-dir artifacts exactly like runs
7
+ started from a plain shell. Fast read-only commands (validate,
8
+ --dry-run) are captured instead, for display in a modal.
9
+
10
+ No textual imports here — everything drives the multiplexer seam (or a plain
11
+ subprocess for the captured read-only commands) and is unit-testable.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import os
17
+ import re
18
+ import stat
19
+ import subprocess
20
+ import sys
21
+ from enum import StrEnum
22
+ from pathlib import Path
23
+
24
+ from .. import runs
25
+ from ..adapters.multiplexer import (
26
+ MultiplexerError,
27
+ get_multiplexer,
28
+ mux_usable,
29
+ )
30
+ from ..journal import Journal
31
+ from ..platform_util import (
32
+ DIR_FD_ANCHORED_WRITES,
33
+ atomic_write_text,
34
+ atomic_write_text_at,
35
+ open_dir_confined,
36
+ )
37
+
38
+ CTL_SESSION = runs.CTL_SESSION
39
+ # The control-session NAME is the transport's business, resolved per call
40
+ # through `runs.ctl_session_for(project)`: the fixed name on tmux, where one
41
+ # server serves the machine and the session really is machine-wide (scoped by
42
+ # the per-window PROJECT_OPTION tag below), and a per-registry name on psmux,
43
+ # whose duplicate-server mutex is keyed on the session name alone, across
44
+ # every registry in the login session (`Local\` is a per-login-session object
45
+ # namespace) — so a fixed name would let only ONE registry there hold a control
46
+ # session and every other project's launch would fail as a duplicate.
47
+ # The constant survives as the fixed base name (display fallbacks, tmux argv
48
+ # pins); anything that addresses a live session resolves the name instead.
49
+
50
+ # control-session windows are named <kind>-<run_id> (see start_detached)
51
+ _CTL_WINDOW_RE = re.compile(r"^(?:run|sweep|resume|resolve)-(.+)$")
52
+
53
+
54
+ class LaunchError(Exception):
55
+ pass
56
+
57
+
58
+ def mux_available() -> bool:
59
+ # Forced-aware (mux_usable, not raw available()): a pinned backend must look
60
+ # the same to observers (attach, ctl-window lookup, prune) as it does to the
61
+ # launch preflight, or a launched run becomes invisible to the rest of the TUI.
62
+ return mux_usable(get_multiplexer())
63
+
64
+
65
+ def session_exists(session: str) -> bool:
66
+ return get_multiplexer().has_session(session)
67
+
68
+
69
+ # Run-dir sidecar naming the ctl-session window start_detached minted last for
70
+ # this run. `<kind>-<run_id>` is not unique across the four kinds, so the window
71
+ # listing alone cannot tell a live resume window from the parked run window it
72
+ # superseded — this file names the one we actually created. A hint, never a
73
+ # target on its own: ctl_window_id re-proves it against the live listing.
74
+ _CTL_WINDOW_FILE = "ctl-window"
75
+
76
+
77
+ # Generous ceiling on the hint: the value is a window id (`@7`, or a
78
+ # session-qualified `froid-loop-ctl:@7`), and anything longer is already not one.
79
+ _MAX_RECORD_BYTES = 256
80
+
81
+
82
+ def _read_ctl_window(project: Path, run_id: str) -> str | None:
83
+ """The window id recorded by the run's last launch, or None when there is
84
+ none / it cannot be read. Never raises, and that includes decoding: a torn
85
+ record can raise UnicodeDecodeError, a ValueError rather than an OSError,
86
+ which action_attach (no covering except at all) and _stop_run_worker (whose
87
+ except does not include it) would let escape. An unreadable hint is not an
88
+ error — it just leaves the caller with the name scan.
89
+
90
+ The file is the only channel on purpose: `froid-loop attach` resolves the same
91
+ run from its own process, and one resolve feeding every consumer is the
92
+ property ctl_window_id sells. A per-process memo of what this process last
93
+ minted would answer a different window than the CLI does.
94
+
95
+ Deliberately not `read_text`. The record sits under the project root every
96
+ coding session can write, and this read runs on Textual's event loop
97
+ (`action_attach` calls it directly), so the *shape* of what is at the path
98
+ has to be established before any bytes are consumed:
99
+
100
+ * `O_NONBLOCK` + an `S_ISREG` check on the opened descriptor. Opening a FIFO
101
+ for reading otherwise blocks until someone writes — indefinitely, freezing
102
+ the dashboard on a keypress.
103
+ * `O_NOFOLLOW`, so the name is read rather than wherever it points.
104
+ * At most `_MAX_RECORD_BYTES`. A record pointed at an endless source reads
105
+ forever otherwise, and it raises `MemoryError` rather than the OSError
106
+ this promises never to leak — `Exception` would catch that but also mask
107
+ real bugs, where a cap removes the condition instead of absorbing it.
108
+
109
+ The check is on the descriptor, not the path, so it cannot be raced: fstat
110
+ describes the object actually opened. The POSIX-only flags degrade to 0 on
111
+ win32, which has neither FIFOs at these paths nor O_NOFOLLOW; the size cap
112
+ and the regular-file check carry there on their own."""
113
+ record = runs.run_dir_for(project, run_id) / _CTL_WINDOW_FILE
114
+ flags = os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0) | getattr(os, "O_NONBLOCK", 0)
115
+ flags |= getattr(os, "O_BINARY", 0) # win32: no CRLF translation on the raw fd
116
+ try:
117
+ fd = os.open(record, flags)
118
+ except OSError:
119
+ return None
120
+ try:
121
+ if not stat.S_ISREG(os.fstat(fd).st_mode):
122
+ return None
123
+ data = os.read(fd, _MAX_RECORD_BYTES)
124
+ except OSError:
125
+ return None
126
+ finally:
127
+ os.close(fd)
128
+ try:
129
+ return data.decode("utf-8").strip() or None
130
+ except UnicodeDecodeError:
131
+ return None
132
+
133
+
134
+ def _forget_ctl_window(project: Path, run_id: str) -> None:
135
+ """Drop the record. A launch that cannot name the window it just minted must
136
+ not leave the *previous* launch's id authoritative — that id now names a
137
+ superseded window, and the honest answer is no record at all, which puts the
138
+ lookup back on the name scan.
139
+
140
+ Ceiling: when the removal fails, or is declined because the path cannot be
141
+ vouched for, the superseded id survives on disk. It still has to pass
142
+ ctl_window_id's re-prove, so the worst it can answer is a live window
143
+ carrying this run's name — the pre-fix by-name result, never a wilder target.
144
+ Retaining a stale hint is strictly the cheaper failure here, which is why
145
+ this declines rather than deleting on a path it cannot stand behind.
146
+
147
+ Anchored exactly like the write in _record_ctl_window, and for a sharper
148
+ reason: a delete needs no race at all. `unlink` does not follow a link at the
149
+ *final* component, but the ancestors resolve normally, so a run dir standing
150
+ as a link to an external directory makes `run_dir / ctl-window` name a file
151
+ over there — another project's live record — and this deletes it. The write
152
+ path's escape needed the attacker to win a window between check and write;
153
+ a planted link just sits there until the next launch fails to capture an id.
154
+ So the descriptor from `open_dir_confined` is what the unlink is relative to,
155
+ and no path is named. win32 keeps the check-then-delete fallback on the same
156
+ terms as the write — see `_run_dir_is_confined` for that residual.
157
+
158
+ A plain unlink, not retrying_unlink: launches run on the Textual event
159
+ loop, and dropping a best-effort hint is not worth ~5s of blocked win32
160
+ backoff — the ceiling above already covers the miss."""
161
+ run_dir = runs.run_dir_for(project, run_id)
162
+ try:
163
+ if DIR_FD_ANCHORED_WRITES:
164
+ dir_fd = open_dir_confined(project, run_dir)
165
+ if dir_fd is None:
166
+ return # a component we cannot vouch for — see the ceiling
167
+ try:
168
+ os.unlink(_CTL_WINDOW_FILE, dir_fd=dir_fd)
169
+ except FileNotFoundError:
170
+ pass # already gone: missing_ok, by hand
171
+ finally:
172
+ os.close(dir_fd)
173
+ else:
174
+ if not _run_dir_is_confined(project, run_dir):
175
+ return # see the ceiling
176
+ (run_dir / _CTL_WINDOW_FILE).unlink(missing_ok=True)
177
+ except OSError:
178
+ pass # a removal we cannot force — see the ceiling
179
+
180
+
181
+ def _is_link_of_any_kind(path: Path) -> bool:
182
+ """Whether `path` is a link that redirects traversal — symlink or, on win32,
183
+ a junction. Raises `OSError` for a component that cannot be probed, which
184
+ the caller turns into a refusal.
185
+
186
+ `is_symlink()` alone is not enough, and the gap is win32-shaped. It answers
187
+ for the symlink reparse tag only and returns **False** for a directory
188
+ junction, which redirects traversal identically. A junction is also the
189
+ *easier* plant of the two: `mklink /J` needs neither elevation nor Developer
190
+ Mode, while a symlink needs one of them. So the check this backs would have
191
+ been blind on win32 to the cheaper version of the very attack it exists for.
192
+
193
+ Detected by the reparse-point attribute rather than `os.path.isjunction`,
194
+ which only exists from 3.12 — this project supports 3.11, and that leg is
195
+ one CI runs on win32. One `lstat`, no version branch: `st_file_attributes`
196
+ is win32-only, so the bit test degrades to False on POSIX, where `S_ISLNK`
197
+ is already the whole answer.
198
+
199
+ Any reparse point counts, not just the junction tag. Other kinds (cloud
200
+ placeholders, app-exec links) have no business being a run dir, and the
201
+ failure this produces is a refusal to write a best-effort hint — the lookup
202
+ degrades to the name scan. Over-refusing is the cheap direction here."""
203
+ info = os.lstat(path)
204
+ if stat.S_ISLNK(info.st_mode):
205
+ return True
206
+ attributes = getattr(info, "st_file_attributes", 0) # win32-only field
207
+ return bool(attributes & stat.FILE_ATTRIBUTE_REPARSE_POINT)
208
+
209
+
210
+ def _run_dir_is_confined(project: Path, run_dir: Path) -> bool:
211
+ """Whether `run_dir` is reached from `project` without traversing a link.
212
+
213
+ `follow_symlinks=False` refuses a link at the *final* component only, which
214
+ leaves the ancestors: a session that replaces `.froid-loop/runs/<run_id>`
215
+ with a link to an external directory holding a `state.json` passes
216
+ `runs.is_run` — it follows the link — and then `mkstemp`/`os.replace` land
217
+ the record inside the linked-to directory. The escape is narrower than the
218
+ final-component one (the name written is always `ctl-window`, so the reach
219
+ is another project's record rather than any file), but it is the same shape.
220
+
221
+ Every component below `project` is checked, and `project` itself is not: the
222
+ operator chooses where the project lives and may well keep it behind a link,
223
+ while everything under it is session-writable. `lstat`-based throughout, so
224
+ the check never resolves through what it is testing for.
225
+
226
+ Each component goes through `_is_link_of_any_kind`, not `is_symlink()` —
227
+ on win32 the latter is blind to a junction, which redirects the same way and
228
+ is the easier of the two to plant. That also fixes a quieter gap: `Path`'s
229
+ predicates swallow the `OSError` from a component that cannot be probed and
230
+ answer False, so an unreadable ancestor used to be walked *past* as "not a
231
+ link" — the opposite of the sentence below. Raising from the probe is what
232
+ makes that sentence true.
233
+
234
+ A check, not a race-free open: the portable answer would be to walk the
235
+ components with `dir_fd`, which POSIX has and win32 does not, and this
236
+ record is atomic precisely for the win32 leg. So the standing redirect —
237
+ plant a link, wait for a launch — is what this removes; a session that
238
+ re-plants inside the window between check and write still wins. That
239
+ residual is bounded by the two facts above: same uid as the writer, and a
240
+ fixed filename carrying a window id."""
241
+ try:
242
+ if not run_dir.is_relative_to(project):
243
+ return False
244
+ cursor = run_dir
245
+ while cursor != project:
246
+ if _is_link_of_any_kind(cursor):
247
+ return False
248
+ cursor = cursor.parent
249
+ except OSError:
250
+ return False # a component we cannot probe is one we cannot vouch for
251
+ return True
252
+
253
+
254
+ def _record_ctl_window(project: Path, run_id: str, win_id: str) -> None:
255
+ """Record the window a launch just minted, so ctl_window_id can prefer it
256
+ over an older window sharing the run id.
257
+
258
+ Best-effort on purpose. The window is already running by the time this
259
+ writes, so a failed write must not fail the launch — the lookup degrades to
260
+ the name scan, i.e. to the behaviour before this record existed. A failure
261
+ forgets the previous record rather than leaving it: degrading to the scan is
262
+ the intended fallback, answering a superseded window is not.
263
+
264
+ Skipped when there is no run yet: a fresh `run`/`sweep` mints the only
265
+ window carrying its run id (nothing to disambiguate), and the run dir is
266
+ created by the detached child — this record deliberately never mkdirs one,
267
+ and must not be written into a run-dir-shaped directory (pruned, partial)
268
+ that runs.is_run reports as not a run.
269
+
270
+ That skip forgets too, and the "nothing to disambiguate" clause above is
271
+ exactly why it must. The clause holds for the case it was written for —
272
+ `new_run_id` mints a fresh id, so no other window carries it — but it does
273
+ not hold for every way of reaching this branch. resume/resolve read state,
274
+ raise a confirm modal, and launch from the callback, so anything that
275
+ removes `state.json` inside that human-length window arrives here with a
276
+ predecessor window live and a previous launch's record still on disk. That
277
+ record names the window this launch just superseded, and ctl_window_id
278
+ prefers any record that still resolves — so `a` and `x` would answer the
279
+ parked predecessor while the orchestrator just minted keeps running, which
280
+ is #482's symptom reintroduced by the record meant to fix it. Dropping it
281
+ puts the lookup back on the name scan, which is this file's stated
282
+ preference throughout: degrading to the scan is the intended fallback,
283
+ answering a superseded window is not.
284
+
285
+ Atomic, not a bare write_text: the record is read cross-process (`froid-loop
286
+ attach`), and on win32 an AV/indexer holding the previous record open fails
287
+ a plain overwrite with a transient sharing violation — which would swallow
288
+ into the forget path and quietly degrade the lookup. atomic_replace retries
289
+ exactly that violation, turning most real-world failures into successes.
290
+
291
+ The guard is type-agnostic on purpose, and `OSError` is not wide enough to
292
+ hold it: `atomic_write_text` resolves the path before its own try, and below
293
+ 3.13 `Path.resolve` reports a symlink loop as `RuntimeError` — which would
294
+ crash the launch this docstring promises to spare, on the interpreters the
295
+ 3.11/3.12 legs run. Same widening, same reason, as the engine's deferred-close
296
+ rollback (`Engine._restore_deferred_closes`). `Exception` and not
297
+ `BaseException`, so a genuine KeyboardInterrupt still gets out.
298
+
299
+ `follow_symlinks=False`, so a symlink at the path is replaced rather than
300
+ written through. Following one is the helper's default contract ("a ledger
301
+ symlinked into the repo keeps being a symlink"), and it is right there — for
302
+ an operator-curated ledger. This sidecar is the opposite: machine-minted,
303
+ per-run, disposable, and living under the project root that every coding
304
+ session can write. Honouring a link here would let a session aim a
305
+ *host-side* write at any path the user can write — reach that the adapters
306
+ confining a session to the workspace otherwise deny it. The payload is only
307
+ a window id, so the primitive is truncation rather than injection, which
308
+ bounds the damage without making it acceptable.
309
+
310
+ Replacing rather than refusing, and no preflight `is_symlink` check: a check
311
+ leaves the window between itself and the write, which a session that
312
+ re-plants the link wins. `os.replace` does not dereference its destination,
313
+ so the link is clobbered whenever it was planted. That also self-heals — the
314
+ record ends up a plain file again — where a refusal would leave the planted
315
+ link in place for the next launch to trip over.
316
+
317
+ Anchored at a directory descriptor where the platform has one. The final
318
+ component is covered by `follow_symlinks=False` above, but the *ancestors*
319
+ are not, and a path check over them (`_run_dir_is_confined`) is answered
320
+ about a path — stale the moment it returns, so a session that re-plants
321
+ `.froid-loop/runs/<run_id>` between check and write still redirects the
322
+ record out of the workspace. `open_dir_confined` walks those components
323
+ `O_NOFOLLOW` and hands back the descriptor for the directory it reached, and
324
+ `atomic_write_text_at` then never names a path again — so a later swap
325
+ renames something this no longer consults, and there is no window to win.
326
+
327
+ win32 keeps the check-then-write path: it has no `*at()` family to anchor
328
+ against (its CPython config defines neither HAVE_RENAMEAT nor HAVE_OPENAT),
329
+ so the descriptor cannot be opened there at all. The residual is documented
330
+ on `_run_dir_is_confined` and bounded by the two facts it names — same uid
331
+ as the writer, and a fixed filename carrying a window id.
332
+ """
333
+ run_dir = runs.run_dir_for(project, run_id)
334
+ if not runs.is_run(run_dir):
335
+ _forget_ctl_window(project, run_id)
336
+ return
337
+ try:
338
+ if DIR_FD_ANCHORED_WRITES:
339
+ dir_fd = open_dir_confined(project, run_dir)
340
+ if dir_fd is None:
341
+ return # unconfined, or a component we cannot vouch for
342
+ try:
343
+ atomic_write_text_at(dir_fd, _CTL_WINDOW_FILE, win_id)
344
+ finally:
345
+ os.close(dir_fd)
346
+ else:
347
+ if not _run_dir_is_confined(project, run_dir):
348
+ return
349
+ atomic_write_text(run_dir / _CTL_WINDOW_FILE, win_id, follow_symlinks=False)
350
+ except Exception:
351
+ _forget_ctl_window(project, run_id)
352
+
353
+
354
+ def ctl_window_id(project: Path, run_id: str) -> str | None:
355
+ """Stable window id (bare `@N` on tmux, session-qualified on psmux) of the
356
+ control-session window hosting this run's orchestrator process
357
+ (start_detached names windows <kind>-<run_id>), or None when the run was
358
+ not launched from the TUI or the session is gone.
359
+
360
+ An id, not a name, because every consumer replays the value as a
361
+ select/kill/option target: one resolve feeds all of them, so a rename or a
362
+ window minted between two verbs cannot send them to different windows, and
363
+ the value survives tmux's automatic-rename.
364
+
365
+ `<kind>-<run_id>` is not unique — a resume launched over a still-parked run
366
+ window shares the run id, and nothing reaps the parked one in between — so
367
+ the name scan alone answers whichever match the listing emits first (tmux
368
+ orders by window *index*, and it gives a new window the lowest free index,
369
+ so a superseded window usually but not always sorts ahead of the live one).
370
+ The id the run's last launch minted is recorded in the run dir and wins
371
+ whenever the listing still shows it under this run id. A record that is gone
372
+ (killed, pruned) or now carries another run's name is ignored rather than
373
+ replayed: a target that no longer resolves is the dangerous kind of stale —
374
+ on psmux an unresolvable `-t` lands on the *active* window (psmux/psmux#545;
375
+ tmux merely errors, which the best-effort consumers turn into a silent
376
+ no-op). With no record at all the answer is the first match, exactly as
377
+ before.
378
+
379
+ Scoped to `project` by the PROJECT_OPTION tag, on the same rule as
380
+ _ctl_window_candidates: the control session is shared across projects, and a
381
+ run id is only unique within one (`--run-id` is caller-supplied), so a
382
+ same-id window belonging to another project would otherwise be a legal match
383
+ here — for `x` that means killing a *live* orchestrator next door. An
384
+ untagged window is admitted when this project has the run dir, which keeps a
385
+ window whose (best-effort) tag write failed reachable by its own project
386
+ rather than by nobody.
387
+
388
+ Untagged is a *fallback*, not a peer: an untagged window proves nothing
389
+ about who owns it, so it is consulted only when nothing carries this
390
+ project's tag. Merged into one listing-ordered list they would compete on
391
+ index, and a neighbouring project's untagged window listed first would beat
392
+ this project's correctly tagged one — for `x`, killing next door's
393
+ orchestrator. That case is not hypothetical: the record cannot break the tie
394
+ for a fresh `run`, where recording is deliberately skipped."""
395
+ if not mux_available():
396
+ return None
397
+ mine = runs.accepted_tags(project)
398
+ local = runs.is_run(runs.run_dir_for(project, run_id))
399
+ tagged: list[str] = []
400
+ untagged: list[str] = []
401
+ rows = get_multiplexer().list_windows(
402
+ ctl_session(project), ["window_id", "window_name", runs.PROJECT_OPTION]
403
+ )
404
+ for win_id, name, tag in rows:
405
+ # win_id can be "": psmux's qualifier passes a falsy id through. An
406
+ # empty id must never become a target — an empty `-t` resolves against
407
+ # the *current* window. (The base's short-row padding CAN produce an
408
+ # empty *tag* — it fills trailing fields — which is exactly the untagged
409
+ # case below; window_id stays field 0 of 3.)
410
+ if not win_id:
411
+ continue
412
+ # The whole run id, not a suffix of the name: RUN_ID_RE admits `-`, so
413
+ # `--run-id other-RID` mints `run-other-RID`, which ends with `-RID` and
414
+ # would answer a lookup for `RID` — and sorts ahead of it, so `x` kills
415
+ # the neighbour's LIVE orchestrator. Parsed with the same regex
416
+ # _ctl_window_candidates uses, which also confines a match to the four
417
+ # kinds start_detached mints rather than any name ending this way.
418
+ m = _CTL_WINDOW_RE.match(name)
419
+ if m is None or m.group(1) != run_id:
420
+ continue
421
+ # Set membership, with no "the tag looks unsafe here" escape: the digest
422
+ # arrives as it was written, so a nonempty tag outside the accepted set
423
+ # belongs to another project and must not be a candidate — `x` resolves
424
+ # through here, and admitting a foreign row lets a stop cross a project
425
+ # boundary. The set is what keeps a window tagged by an earlier release
426
+ # reachable: the control session is long-lived and survives the upgrade
427
+ # that changes the tag's spelling, so comparing against the current
428
+ # digest alone would strand this project's own orchestrator — prunable
429
+ # by _ctl_window_candidates, which accepts the legacy tag, yet
430
+ # unreachable by `a` and `x`, which resolve through here.
431
+ if tag in mine:
432
+ tagged.append(win_id)
433
+ elif not tag and local:
434
+ # untagged, and this project holds the run dir — ownership is
435
+ # plausible but unproven, so it only counts if nothing is tagged
436
+ untagged.append(win_id)
437
+ matches = tagged or untagged
438
+ if not matches:
439
+ return None
440
+ # Membership in `matches`, not mere presence in the listing: it re-proves the
441
+ # name and the project too, so neither a backend that reuses a freed window
442
+ # id nor a record naming a neighbouring project's window can be replayed.
443
+ recorded = _read_ctl_window(project, run_id)
444
+ return recorded if recorded in matches else matches[0]
445
+
446
+
447
+ def ctl_window_recorded(project: Path, run_id: str, win_id: str) -> bool:
448
+ """Whether `ctl_window_id` now answers `win_id` for this run — i.e. whether
449
+ the launch's disambiguation actually took.
450
+
451
+ False means the launch itself succeeded but the lookup is back on the
452
+ ambiguous first-match scan, which is exactly #482's symptom and so is
453
+ operator-visible: every launcher that mints a second window under a run id
454
+ should report it rather than let an unqualified success toast imply the
455
+ targeting is sound. Split out of resume_detached's return so the resolve
456
+ path can warn while still keeping the captured id it attaches with.
457
+
458
+ Asks `ctl_window_id` rather than comparing the record to `win_id`, because
459
+ a round-tripped record is not the same claim. A backend whose
460
+ `new_parked_window` id is shaped differently from its `list_windows`
461
+ `window_id` column — a divergence the seam explicitly tolerates — writes and
462
+ reads the record back intact while `ctl_window_id` rejects it against the
463
+ listing and falls through to the first match. File equality would report
464
+ that as sound; it is the precise case the warning exists for.
465
+
466
+ An unanswerable listing counts as not recorded. The probe is observation, so
467
+ it degrades rather than raising into the launchers (neither has a handler
468
+ for it), and "could not confirm" is closer to the warning's own hedge —
469
+ attach/stop *may* target an older window — than silence would be."""
470
+ try:
471
+ return ctl_window_id(project, run_id) == win_id
472
+ except MultiplexerError:
473
+ return False
474
+
475
+
476
+ def ctl_target(project: Path) -> str:
477
+ """Seam-canonical target token for this project's control session; see
478
+ :meth:`TerminalMultiplexer.target`. Windows are targeted by stable id
479
+ (ctl_window_id), never by name through this token."""
480
+ return get_multiplexer().target(ctl_session(project))
481
+
482
+
483
+ def select_ctl_window_id(window_id: str) -> None:
484
+ """Make the window with this id (from start_detached/ctl_window_id) the
485
+ control session's current window, so a plain attach to the session lands
486
+ on it (attach-session itself takes no window)."""
487
+ get_multiplexer().select_window(window_id)
488
+
489
+
490
+ # Per-window tmux user option recording what an interactive attach should do
491
+ # with the client once the window's command exits (consumed by the multiplexer's
492
+ # parked-window return trailer; see start_detached and the tmux backend). Set by
493
+ # set_return_pane at attach time. Value is either a backend-composed pane
494
+ # target — replayed opaquely, so each backend records the form its own
495
+ # switch-client resolves: a bare pane id (%N) on tmux, =session:%N on psmux,
496
+ # whose one-server-per-session model cannot resolve a bare id from the control
497
+ # session (psmux/psmux#483) — used when the TUI runs inside the multiplexer and
498
+ # switched its own client over; or RETURN_DETACH, used when the TUI runs
499
+ # outside and a throwaway client was attached that must detach so the
500
+ # suspended TUI resumes.
501
+ RETURN_OPTION = "@froid_return_pane"
502
+ RETURN_DETACH = "detach" # pane targets are %N / =sess:%N, never "detach"
503
+
504
+
505
+ def current_return_target() -> str | None:
506
+ """Backend-composed target of the pane this process runs in — the place an
507
+ attach should return the client to — or None when not inside the
508
+ multiplexer / it is unavailable. The value is opaque to callers: record it
509
+ with set_return_pane, replay it via switch_client / the parked trailer.
510
+ See TerminalMultiplexer.current_return_target for the composition
511
+ contract."""
512
+ return get_multiplexer().current_return_target()
513
+
514
+
515
+ def set_return_pane(window_target: str, target: str) -> None:
516
+ """Record `target` (a current_return_target value or RETURN_DETACH) as the
517
+ return move on a control-session window, so its trailing shell sends the
518
+ client back there when the window's command exits. `window_target` is any
519
+ window spec the backend accepts; callers pass the id from
520
+ start_detached/ctl_window_id so the write lands on the window the caller
521
+ already resolved, not on whatever a fresh by-name lookup answers."""
522
+ get_multiplexer().set_window_option(window_target, RETURN_OPTION, target)
523
+
524
+
525
+ def current_session() -> str | None:
526
+ """Name of the tmux session this process is running inside, or None when
527
+ not in tmux / tmux is unavailable."""
528
+ return get_multiplexer().current_session()
529
+
530
+
531
+ def in_ctl_session() -> bool:
532
+ """True when we are running inside a control-session window (i.e. launched
533
+ detached by the TUI), as opposed to a user's own shell. Backend-honest:
534
+ current_session() is None whenever this process is not inside the selected
535
+ multiplexer, so no direct TMUX/HERDR_* env sniffing happens here. The
536
+ shape predicate rather than one project's resolved name: the question its
537
+ callers ask is "am I in A control session", and on a namespacing transport
538
+ the name carries a registry suffix (runs.ctl_session_for)."""
539
+ session = current_session()
540
+ return session is not None and runs.is_ctl_session_name(session)
541
+
542
+
543
+ def detach_client() -> bool:
544
+ """Detach the tmux client viewing the current session, handing the terminal
545
+ back to the user. Processes in the session keep running. Returns True iff a
546
+ client was actually detached — False both when the transport failed and when
547
+ there was nothing attached (see TerminalMultiplexer.detach_client for how
548
+ each backend establishes that)."""
549
+ return get_multiplexer().detach_client()
550
+
551
+
552
+ class ReturnOutcome(StrEnum):
553
+ """What return_attached_client managed to do — and, for a caller that goes
554
+ unattended on the strength of it, whether a human can still answer here.
555
+
556
+ A plain boolean cannot carry that: "the hand-back succeeded" and "there is
557
+ still someone at this terminal" are independent, and the ways of failing
558
+ point in different directions. A *refused* switch leaves the client sitting
559
+ in this very window; a switch the backend cannot vouch for may already have
560
+ moved it; a failed *detach* reports no verified hand-back. Three claims, and
561
+ they do not license the same response."""
562
+
563
+ RETURNED = "returned"
564
+ #: No hand-back, but a human may still be here: nothing was recorded to
565
+ #: return to (a plain foreground sweep), the backend is unusable, or the
566
+ #: switch failed with the client still in this window. The conservative
567
+ #: answer — a caller must keep talking to the terminal.
568
+ ATTENDED = "attended"
569
+ #: A hand-back was attempted and did not verifiably happen: the detach found
570
+ #: nothing attached, the switch could not be vouched for (a timed-out verb,
571
+ #: an unreadable client count, nothing attached to move), the effect could
572
+ #: not be observed, or the backend has no detach verb at all (herdr). A
573
+ #: caller must not rely on anyone answering a prompt in this window — a
574
+ #: policy for the uncertainty, not a proof that the window is empty (see
575
+ #: return_attached_client for why it is the safe way to be wrong).
576
+ UNREACHABLE = "unreachable"
577
+
578
+
579
+ def return_attached_client() -> ReturnOutcome:
580
+ """Hand an attached client back to its origin *now*, mid-process — the
581
+ parked-window return move (see start_detached) executed while the window's
582
+ command keeps running in the background, instead of after it exits.
583
+
584
+ Reads the RETURN_OPTION recorded on the current window by set_return_pane:
585
+ - a pane target (backend-composed: bare %N on tmux, =session:%N on
586
+ psmux): switch that client back there (`-l` fallback if it's gone);
587
+ - RETURN_DETACH: detach the client so a blocking `tmux attach` returns;
588
+ - unset/empty: nobody attached with a return target — do nothing.
589
+ The option is cleared only on RETURNED: a real return must not make the
590
+ parked window's trailer fire a second one, a failed return is left for the
591
+ trailer to retry. That retry is a second chance, not a rescue —
592
+ new_parked_window parks on a blocking read *before* the trailer, so it runs
593
+ only once a human dismisses the park prompt, never in the unattended case.
594
+
595
+ The two failures are not interchangeable, which is why this answers a
596
+ ReturnOutcome and not a bool. A failed switch is positive evidence that the
597
+ client is still in this window, so ATTENDED keeps the caller prompting —
598
+ but only because the seam reserves False for that joint claim. A backend
599
+ that merely cannot vouch for the move (psmux's timed-out verb, an
600
+ unreadable client count, nothing attached to move) answers None, and that
601
+ lands in UNREACHABLE instead. The routing is load-bearing, not tidiness:
602
+ an ATTENDED the client has already walked away from cannot be recovered by
603
+ the surviving return option, because a --repeat cycle prompting into the
604
+ empty window blocks on input() before anyone can reach the trailer. A
605
+ failed detach carries no such evidence in general: on tmux it does
606
+ (`detach-client` fails with "no current client"), but off tmux False also
607
+ covers an effect the backend could not observe and a detach verb it does
608
+ not have at all — herdr, whose False rather than None is exactly what
609
+ detach_client's own widening to a returned bool (#317) buys. That is a
610
+ different widening from switch_client's third state above; detach_client is
611
+ the verb that stays a bool. UNREACHABLE is the policy for all of them, the
612
+ unvouched switch included, because the two ways of being wrong are not
613
+ equally bad: prompting into a
614
+ window no one is viewing blocks a --repeat sweep on input() forever, while
615
+ going unattended in front of a human only defers this cycle's decisions to
616
+ `froid-loop decisions` or the next attended sweep."""
617
+ mux = get_multiplexer()
618
+ if not mux_usable(mux):
619
+ return ReturnOutcome.ATTENDED
620
+ win = mux.current_window_id()
621
+ if win is None:
622
+ return ReturnOutcome.ATTENDED
623
+ ret = mux.show_window_option(win, RETURN_OPTION)
624
+ if not ret:
625
+ return ReturnOutcome.ATTENDED
626
+ if ret == RETURN_DETACH:
627
+ outcome = ReturnOutcome.RETURNED if mux.detach_client() else ReturnOutcome.UNREACHABLE
628
+ else:
629
+ switched = mux.switch_client(ret, last_fallback=True)
630
+ if switched is None:
631
+ outcome = ReturnOutcome.UNREACHABLE
632
+ else:
633
+ outcome = ReturnOutcome.RETURNED if switched else ReturnOutcome.ATTENDED
634
+ if outcome is ReturnOutcome.RETURNED:
635
+ mux.unset_window_option(win, RETURN_OPTION)
636
+ return outcome
637
+
638
+
639
+ def decision_pending(run_dir: Path) -> bool:
640
+ """True when the run's sweep is currently blocked on an interactive decision
641
+ — its journal's last entry is a decision-pending announcement (the prompter
642
+ blocks on input right after writing it, so any later entry means it moved
643
+ on). Mirrors tui.data.pending_decision; kept here so the CLI can decide an
644
+ attach target without importing the textual-laden data module."""
645
+ entries = Journal(run_dir).entries()
646
+ return bool(entries) and entries[-1].get("kind") == "decision-pending"
647
+
648
+
649
+ def attach_plan(project: Path, run_id: str) -> tuple[list[str], str | None] | None:
650
+ """Pick where an interactive attach should land for this run and which window
651
+ (if any) to record a return target on. Shared by the CLI `attach` command and
652
+ mirroring the TUI's action_attach logic: prefer the orchestrator's ctl window
653
+ when a sweep is blocked on a decision or no agent session is live, else the
654
+ live agent session. Returns (tmux argv, return_window) or None when there is
655
+ nothing to attach to."""
656
+ session = runs.session_name(run_id)
657
+ win_id = ctl_window_id(project, run_id)
658
+ agent_live = session_exists(session)
659
+ if win_id is not None and (
660
+ decision_pending(runs.run_dir_for(project, run_id)) or not agent_live
661
+ ):
662
+ select_ctl_window_id(win_id)
663
+ return runs.attach_target_argv(ctl_target(project)), win_id
664
+ if agent_live:
665
+ return runs.attach_target_argv(runs.session_target(run_id)), None
666
+ return None
667
+
668
+
669
+ def kill_ctl_window(project: Path, run_id: str) -> None:
670
+ """Kill the control-session window hosting this run's orchestrator process,
671
+ if any. A no-op when the run was not launched from the TUI or tmux is gone."""
672
+ win_id = ctl_window_id(project, run_id)
673
+ if win_id is not None:
674
+ get_multiplexer().kill_window(win_id)
675
+
676
+
677
+ def _ctl_window_candidates(project: Path) -> list[tuple[str, str]]:
678
+ """(window_id, window_name) for parked control-session run windows whose run
679
+ is no longer live — the kill candidates for a prune.
680
+
681
+ A `<kind>-<run_id>` window parks on a `read` prompt that never closes on its
682
+ own; it is a candidate once its run has finished/stopped/crashed (or its run
683
+ dir is gone). The current window is excluded so a prune triggered from inside
684
+ the ctl session never targets itself; live runs and the session's own shell
685
+ window are excluded too.
686
+
687
+ The control session is shared across projects, so its per-window PROJECT_OPTION
688
+ accepts current and legacy project tags; untagged windows still require a run
689
+ directory under this project (mirrors runs.prunable_sessions).
690
+ """
691
+ mux = get_multiplexer()
692
+ ctl = runs.ctl_session_for(project, mux)
693
+ if not mux_usable(mux) or not session_exists(ctl):
694
+ return []
695
+ current = mux.current_window_id()
696
+ rows = mux.list_windows(ctl, ["window_id", "window_name", runs.PROJECT_OPTION])
697
+ mine = runs.accepted_tags(project)
698
+ candidates: list[tuple[str, str]] = []
699
+ for win_id, name, tag in rows:
700
+ if not win_id or win_id == current:
701
+ continue
702
+ m = _CTL_WINDOW_RE.match(name)
703
+ if m is None:
704
+ continue # not a run window (e.g. the session's initial shell)
705
+ if not runs.is_parsable_run_id(m.group(1)):
706
+ # A foreign/mangled window name must not steer a run-dir path. The
707
+ # PARSE-side predicate: this window already exists, so the mint's
708
+ # broad ctl reservation would leak every pre-upgrade `run-ctl-*`
709
+ # window out of the sweep instead of closing it.
710
+ continue
711
+ run_dir = runs.run_dir_for(project, m.group(1))
712
+ if tag:
713
+ if tag not in mine:
714
+ continue # another project's window
715
+ elif not runs.is_run(run_dir):
716
+ continue # untagged and no run dir here — ownership unprovable
717
+ # boolean gate on purpose: an 'unknown' engine stays a candidate (unknown
718
+ # never blocks cleanup) with no per-window warning — the session-level
719
+ # unknown warning from prunable_sessions covers the operator surface.
720
+ if runs.engine_alive(run_dir):
721
+ continue
722
+ candidates.append((win_id, name))
723
+ return candidates
724
+
725
+
726
+ def prunable_ctl_windows(project: Path) -> list[str]:
727
+ """Names of the control-session windows a prune would close (dry-run view)."""
728
+ return [name for _, name in _ctl_window_candidates(project)]
729
+
730
+
731
+ def prune_ctl_windows(project: Path) -> tuple[list[str], list[str], list[str]]:
732
+ """Close parked control-session windows whose run is no longer live; returns
733
+ (removed, survived, unverifiable) window names (see _ctl_window_candidates).
734
+ A three-list tuple like runs.prune_sessions, but do NOT read the arms across:
735
+ that one partitions BEFORE its kills, so its `killed` is still an attempted
736
+ kill, its `live` is "deliberately not touched" rather than "survived", and its
737
+ `unknown` is a pid question and a SUBSET of `killed`. These three are disjoint
738
+ (by window id — the values are names) and all three are about kill outcome.
739
+
740
+ kill_window is best-effort by contract (a hang, a missing binary, and a
741
+ refused kill are all the same silent no-op), so an attempted kill is not a
742
+ removal and must not be reported as one (#435). The verdict is taken here
743
+ rather than pushed into the seam because this is the caller that both needs
744
+ it and already holds the session: kill_window(target) alone cannot verify
745
+ anything on a backend whose liveness listing is session-scoped, which is all
746
+ of them.
747
+
748
+ ONE listing after the whole fan-out, not a probe per window: the answer is a
749
+ set membership either way, so the verdict costs one extra round trip instead
750
+ of N. A transport fault raises and nothing can be claimed there.
751
+
752
+ Two ceilings, both deliberate:
753
+
754
+ - The membership test pairs list_windows' `window_id` column with
755
+ list_window_ids. The seam states its symmetry rules pairwise and this pair
756
+ is stated because of THIS caller — a backend qualifying one side and not
757
+ the other reads every candidate as removed, which is #435 restored on the
758
+ optimistic side, with no error anywhere.
759
+ - `[]` is read as "the session went with its last window". The seam's `[]`
760
+ is wider than that: BaseTmuxBackend folds EVERY nonzero exit to `[]`, so a
761
+ server that errors while its windows live would report them removed. The
762
+ common cause by far is the session really being gone, and pessimism there
763
+ would invent a phantom survivor on every future sweep. Narrowing the
764
+ sentinel is a change to the engine's liveness probe, not to this function.
765
+ """
766
+ mux = get_multiplexer()
767
+ candidates = _ctl_window_candidates(project)
768
+ if not candidates:
769
+ return [], [], []
770
+ for win_id, _name in candidates:
771
+ # kill_window is best-effort and reports nothing; a strict-POSIX decode
772
+ # fault of the kill's own capture escapes its swallow tuple (#380) but
773
+ # says exactly as little about the outcome — the command may well have
774
+ # reached the server. More of the same nothing: the fan-out continues
775
+ # and the one post-kill listing hands down the verdict either way. An
776
+ # escape here would surface at the callers as a scan failure, an
777
+ # empty-armed receipt denying kills that just fired.
778
+ try:
779
+ mux.kill_window(win_id)
780
+ except UnicodeError:
781
+ pass
782
+ try:
783
+ live = set(mux.list_window_ids(runs.ctl_session_for(project, mux)))
784
+ except MultiplexerError:
785
+ # The kills may well have landed; nothing here can say so. Claiming the
786
+ # optimistic half is exactly the bug — the next cleanup pass retries.
787
+ return [], [], [name for _win_id, name in candidates]
788
+ removed = [name for win_id, name in candidates if win_id not in live]
789
+ survived = [name for win_id, name in candidates if win_id in live]
790
+ return removed, survived, []
791
+
792
+
793
+ def ctl_session(project: Path) -> str:
794
+ """The control-session name for this project on the selected transport
795
+ (see `runs.ctl_session_for`). The TUI-facing spelling, so `tui/app.py`
796
+ and the screens name the session an operator would actually attach to."""
797
+ return runs.ctl_session_for(project, get_multiplexer())
798
+
799
+
800
+ def _ensure_ctl_session(project: Path) -> str:
801
+ mux = get_multiplexer()
802
+ name = runs.ctl_session_for(project, mux)
803
+ # has_session is raiser-side (a server-backed backend can fail the probe after
804
+ # the availability pre-gate). Keep it inside the try so a transport failure
805
+ # converts to LaunchError, which the TUI launch/resume/resolve handlers already
806
+ # catch — otherwise the raw MultiplexerError slips past them and crashes the app.
807
+ try:
808
+ if not mux.has_session(name):
809
+ mux.new_session(name, project)
810
+ except MultiplexerError as e:
811
+ raise LaunchError(f"multiplexer ctl-session setup failed: {e}") from e
812
+ return name
813
+
814
+
815
+ def cli_argv(*tail: str) -> list[str]:
816
+ """`sys.executable -m froid_loop.cli ...` — immune to PATH/venv drift
817
+ inside tmux windows."""
818
+ return [sys.executable, "-m", "froid_loop.cli", *tail]
819
+
820
+
821
+ def start_detached(project: Path, argv_tail: list[str], run_id: str, kind: str) -> str | None:
822
+ """Run a froid-loop command in a new window of the control session.
823
+
824
+ The window parks after the command exits (keeping the exit status
825
+ inspectable) and then returns an attached client to its origin pane — both
826
+ handled by the multiplexer's parked-window primitive, keyed by the
827
+ RETURN_OPTION recorded on the window by set_return_pane.
828
+
829
+ Returns the new window's stable backend id (bare `@N` on tmux,
830
+ session-qualified on psmux) so callers can target it unambiguously (window
831
+ names collide when several kinds share a run_id). The same id is recorded in
832
+ the run dir so ctl_window_id answers this window rather than an older one
833
+ under the same run id — see _record_ctl_window.
834
+
835
+ Refuses a run id that aliases a control session, FIRST — this is the one
836
+ place every drive path converges on the mutation (the window mint and the
837
+ ctl-window record overwrite): run/sweep launches with freshly validated
838
+ ids, and resume/resolve replaying ids an older release persisted. Gating
839
+ each button separately kept finding the path nobody gated (resolve was
840
+ the fourth); gating the mutation cannot. Ahead of the mux probes so the
841
+ refusal needs no transport to be phrased.
842
+ """
843
+ if runs.run_id_aliases_control_session(run_id):
844
+ raise LaunchError(
845
+ f"run {run_id}: its agent session name is the control session's own — "
846
+ f"cannot be driven. Recover its work by hand, then `froid-loop delete {run_id}`"
847
+ )
848
+ mux = get_multiplexer()
849
+ if not mux_usable(mux):
850
+ raise LaunchError(
851
+ "multiplexer backend unavailable (binary missing, version unsupported, "
852
+ "or a required helper absent)"
853
+ )
854
+ ctl = _ensure_ctl_session(project)
855
+ try:
856
+ win_id = (
857
+ mux.new_parked_window(
858
+ ctl,
859
+ f"{kind}-{run_id}",
860
+ project,
861
+ cli_argv(*argv_tail),
862
+ RETURN_OPTION,
863
+ )
864
+ or None
865
+ )
866
+ except MultiplexerError as e:
867
+ raise LaunchError(f"multiplexer new-window failed: {e}") from e
868
+ if win_id:
869
+ # Record before tagging: a window minted but unrecorded puts the lookup
870
+ # back on the ambiguous scan, while an *untagged* window already has a
871
+ # documented fallback in _ctl_window_candidates — so even a
872
+ # non-conforming backend raising from the (contractually best-effort)
873
+ # set_window_option must not cost the record.
874
+ _record_ctl_window(project, run_id, win_id)
875
+ # Tag the window with its project so a cleanup in another project never
876
+ # closes it (the ctl session is shared across projects).
877
+ mux.set_window_option(win_id, runs.PROJECT_OPTION, runs.project_tag(project))
878
+ else:
879
+ # No id to record: the backend did not capture one. Whatever the previous
880
+ # launch recorded now names a superseded window, so drop it.
881
+ _forget_ctl_window(project, run_id)
882
+ return win_id
883
+
884
+
885
+ def start_run_detached(
886
+ project: Path,
887
+ run_id: str,
888
+ *,
889
+ spec: str | None = None,
890
+ epic: int | None = None,
891
+ story: str | None = None,
892
+ max_stories: int | None = None,
893
+ ) -> None:
894
+ tail = ["run", "--project", str(project), "--run-id", run_id]
895
+ if spec:
896
+ tail += ["--spec", spec] # forces stories mode (folder+id dispatch)
897
+ if epic is not None:
898
+ tail += ["--epic", str(epic)]
899
+ if story:
900
+ tail += ["--story", story]
901
+ if max_stories is not None:
902
+ tail += ["--max-stories", str(max_stories)]
903
+ start_detached(project, tail, run_id, "run")
904
+
905
+
906
+ def start_sweep_detached(
907
+ project: Path,
908
+ run_id: str,
909
+ *,
910
+ no_prompt: bool = False,
911
+ decisions_only: bool = False,
912
+ max_bundles: int | None = None,
913
+ ) -> None:
914
+ tail = ["sweep", "--project", str(project), "--run-id", run_id]
915
+ if no_prompt:
916
+ tail.append("--no-prompt")
917
+ if decisions_only:
918
+ tail.append("--decisions-only")
919
+ if max_bundles is not None:
920
+ tail += ["--max-bundles", str(max_bundles)]
921
+ start_detached(project, tail, run_id, "sweep")
922
+
923
+
924
+ def resume_detached(project: Path, run_id: str) -> str | None:
925
+ """Resume in a ctl-session window; returns the window id, or None when the
926
+ lookup cannot name that window afterwards — the caller should warn, because
927
+ resume is the launch that mints a *second* window under the run id, so this
928
+ is exactly when the ambiguous scan starts answering the superseded one while
929
+ the launch itself succeeded.
930
+
931
+ Two ways to land there, one signal: the backend captured no id, or it did but
932
+ the record did not survive (refused, unwritable, run dir pruned mid-launch).
933
+ Both leave `ctl_window_id` on the scan, so reporting only the first would let
934
+ the rest degrade behind an unqualified success toast.
935
+
936
+ Verified by re-reading rather than by threading the write's outcome up: it
937
+ asks the question the consumers actually ask — will `ctl_window_id` prefer
938
+ this window — of the same file they will read, instead of a proxy for it.
939
+
940
+ Folded into the return here, rather than reported alongside the id as the
941
+ resolve path does, because resume has no immediate use for a window it
942
+ cannot record: it launches and leaves, where resolve attaches to the window
943
+ it just minted and still needs that id to do so."""
944
+ win_id = start_detached(
945
+ project, ["resume", "--project", str(project), run_id], run_id, "resume"
946
+ )
947
+ if win_id and not ctl_window_recorded(project, run_id, win_id):
948
+ return None
949
+ return win_id
950
+
951
+
952
+ def start_resolve_detached(project: Path, run_id: str) -> str | None:
953
+ """Run `froid-loop resolve <run_id>` in a ctl-session window. The caller
954
+ attaches to it: the resolve agent is interactive, and the post-session
955
+ confirm + resume happen in that same window. Returns the window id so the
956
+ caller attaches to exactly this window, not a stale same-run_id window."""
957
+ return start_detached(
958
+ project, ["resolve", "--project", str(project), run_id], run_id, "resolve"
959
+ )
960
+
961
+
962
+ def run_captured_streams(argv_tail: list[str]) -> tuple[int, str, str]:
963
+ """Run a fast read-only command (validate, --dry-run) and capture its output
964
+ with the two streams kept **apart**.
965
+
966
+ Separation is the whole point of this seam. Anything parsing stdout as a
967
+ whole document — a ``--json`` command, whose contract in :mod:`froid_loop.machine`
968
+ is that stdout is one JSON object and nothing else — cannot use the merged
969
+ form: :func:`run_captured` appends stderr *after* stdout, so a single
970
+ ``DeprecationWarning`` written to the child's stderr by any dependency turns
971
+ ``json.loads`` into ``Extra data:``. That failure is environment-dependent —
972
+ it needs the right interpreter, the right installed versions, the right
973
+ warning filters — so it would pass everywhere it was tested and silently
974
+ degrade the JSON path to the text one on a user's machine. A caller that
975
+ genuinely wants one blob merges them itself; a caller that parses must never
976
+ have been handed the option.
977
+
978
+ Decoding is pinned to UTF-8 with ``errors="replace"`` rather than
979
+ ``text=True``, which decodes with the *locale* encoding at ``errors="strict"``
980
+ — the #200 family of failure already fixed CLI-side in :mod:`froid_loop.machine`.
981
+ A console in a non-UTF-8 code page must not turn a perfectly good document
982
+ into a ``UnicodeDecodeError`` on the way in.
983
+ """
984
+ proc = subprocess.run(
985
+ cli_argv(*argv_tail), capture_output=True, encoding="utf-8", errors="replace"
986
+ )
987
+ return proc.returncode, proc.stdout, proc.stderr
988
+
989
+
990
+ def run_captured(argv_tail: list[str]) -> tuple[int, str]:
991
+ """Run a fast read-only command (validate, --dry-run) and capture its
992
+ combined output for display.
993
+
994
+ For text display only. Anything that parses the output must call
995
+ :func:`run_captured_streams` — see its docstring on why the merge is
996
+ unparseable.
997
+ """
998
+ rc, out, err = run_captured_streams(argv_tail)
999
+ if err:
1000
+ if out and not out.endswith("\n"):
1001
+ out += "\n"
1002
+ out += err
1003
+ return rc, out