froid-loop 0.11.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. froid_loop/__init__.py +11 -0
  2. froid_loop/__main__.py +12 -0
  3. froid_loop/adapters/__init__.py +3 -0
  4. froid_loop/adapters/base.py +254 -0
  5. froid_loop/adapters/entrypoints.py +63 -0
  6. froid_loop/adapters/env_fault.py +290 -0
  7. froid_loop/adapters/generic.py +2013 -0
  8. froid_loop/adapters/mock.py +49 -0
  9. froid_loop/adapters/multiplexer.py +914 -0
  10. froid_loop/adapters/opencode_http.py +1687 -0
  11. froid_loop/adapters/profile.py +650 -0
  12. froid_loop/adapters/psmux_backend.py +1428 -0
  13. froid_loop/adapters/registry.py +322 -0
  14. froid_loop/adapters/tmux_backend.py +35 -0
  15. froid_loop/adapters/tmux_base.py +630 -0
  16. froid_loop/checks.py +187 -0
  17. froid_loop/cli.py +5041 -0
  18. froid_loop/data/__init__.py +0 -0
  19. froid_loop/data/froid_loop_hook.py +228 -0
  20. froid_loop/data/froid_loop_probe_hook.py +88 -0
  21. froid_loop/data/plugins/example/plugin.toml +21 -0
  22. froid_loop/data/plugins/tea/plugin.toml +184 -0
  23. froid_loop/data/plugins/tea/tea_plugin.py +258 -0
  24. froid_loop/data/plugins/unity/plugin.toml +140 -0
  25. froid_loop/data/plugins/unity/unity_assets/FroidLoop.Unity.Editor.asmdef +16 -0
  26. froid_loop/data/plugins/unity/unity_assets/FroidLoop.Unity.Editor.asmdef.meta +7 -0
  27. froid_loop/data/plugins/unity/unity_assets/SceneAutoSaveGuard.cs +221 -0
  28. froid_loop/data/plugins/unity/unity_assets/SceneAutoSaveGuard.cs.meta +11 -0
  29. froid_loop/data/plugins/unity/unity_assets/_folders/Editor.meta +8 -0
  30. froid_loop/data/plugins/unity/unity_assets/_folders/FroidLoop.meta +8 -0
  31. froid_loop/data/plugins/unity/unity_cleanup.py +125 -0
  32. froid_loop/data/plugins/unity/unity_dialog_probe.py +239 -0
  33. froid_loop/data/plugins/unity/unity_facts.md +17 -0
  34. froid_loop/data/plugins/unity/unity_plugin.py +415 -0
  35. froid_loop/data/plugins/unity/unity_quiesce.py +234 -0
  36. froid_loop/data/plugins/unity/unity_ready.py +230 -0
  37. froid_loop/data/plugins/unity/unity_seed_assets.py +298 -0
  38. froid_loop/data/plugins/unity/unity_setup.py +551 -0
  39. froid_loop/data/plugins/unity/unity_teardown.py +362 -0
  40. froid_loop/data/profiles/antigravity.toml +52 -0
  41. froid_loop/data/profiles/claude.toml +85 -0
  42. froid_loop/data/profiles/codex.toml +22 -0
  43. froid_loop/data/profiles/copilot.toml +52 -0
  44. froid_loop/data/profiles/gemini.toml +26 -0
  45. froid_loop/data/profiles/opencode.toml +54 -0
  46. froid_loop/data/settings/core.toml +458 -0
  47. froid_loop/data/skills/README.md +93 -0
  48. froid_loop/data/skills/froid-loop-resolve/SKILL.md +288 -0
  49. froid_loop/data/skills/froid-loop-setup/SKILL.md +161 -0
  50. froid_loop/data/skills/froid-loop-setup/assets/module-help.csv +3 -0
  51. froid_loop/data/skills/froid-loop-setup/assets/module.yaml +19 -0
  52. froid_loop/data/skills/froid-loop-sweep/SKILL.md +100 -0
  53. froid_loop/data/skills/froid-loop-sweep/automation-mode.md +127 -0
  54. froid_loop/data/skills/froid-loop-sweep/deferred-work-format.md +302 -0
  55. froid_loop/data/skills/froid-loop-sweep/migration-mode.md +86 -0
  56. froid_loop/decisions.py +202 -0
  57. froid_loop/deferredwork.py +2282 -0
  58. froid_loop/devcontract.py +892 -0
  59. froid_loop/diagnostics.py +1104 -0
  60. froid_loop/documents.py +532 -0
  61. froid_loop/engine.py +7732 -0
  62. froid_loop/envvars.py +111 -0
  63. froid_loop/escalation.py +225 -0
  64. froid_loop/events.py +266 -0
  65. froid_loop/fences.py +103 -0
  66. froid_loop/froidconfig.py +226 -0
  67. froid_loop/frontmatter.py +526 -0
  68. froid_loop/gates.py +133 -0
  69. froid_loop/install.py +2936 -0
  70. froid_loop/journal.py +178 -0
  71. froid_loop/machine.py +148 -0
  72. froid_loop/model.py +898 -0
  73. froid_loop/operatoractions.py +474 -0
  74. froid_loop/platform_util.py +1490 -0
  75. froid_loop/plugins/__init__.py +64 -0
  76. froid_loop/plugins/bus.py +259 -0
  77. froid_loop/plugins/context.py +319 -0
  78. froid_loop/plugins/loader.py +145 -0
  79. froid_loop/plugins/manifest.py +279 -0
  80. froid_loop/plugins/model.py +296 -0
  81. froid_loop/plugins/registry.py +245 -0
  82. froid_loop/plugins/trust.py +75 -0
  83. froid_loop/policy.py +1569 -0
  84. froid_loop/probe.py +1044 -0
  85. froid_loop/process_host.py +408 -0
  86. froid_loop/recovery_flow.py +1561 -0
  87. froid_loop/resolve.py +283 -0
  88. froid_loop/runs.py +4715 -0
  89. froid_loop/runsetup.py +1293 -0
  90. froid_loop/sanitize.py +593 -0
  91. froid_loop/settings_schema.py +276 -0
  92. froid_loop/signals.py +160 -0
  93. froid_loop/sprintstatus.py +609 -0
  94. froid_loop/statemachine.py +57 -0
  95. froid_loop/stories.py +615 -0
  96. froid_loop/stories_engine.py +796 -0
  97. froid_loop/sweep.py +1892 -0
  98. froid_loop/tokens.py +196 -0
  99. froid_loop/tui/__init__.py +11 -0
  100. froid_loop/tui/app.py +1584 -0
  101. froid_loop/tui/data.py +840 -0
  102. froid_loop/tui/launch.py +1003 -0
  103. froid_loop/tui/screens/__init__.py +1 -0
  104. froid_loop/tui/screens/dashboard.py +1071 -0
  105. froid_loop/tui/screens/modals.py +943 -0
  106. froid_loop/tui/screens/settings_screen.py +477 -0
  107. froid_loop/tui/settings.py +135 -0
  108. froid_loop/tui/widgets.py +981 -0
  109. froid_loop/verify.py +4545 -0
  110. froid_loop/workspace.py +320 -0
  111. froid_loop/worktree_flow.py +2301 -0
  112. froid_loop-0.11.1.dist-info/METADATA +728 -0
  113. froid_loop-0.11.1.dist-info/RECORD +116 -0
  114. froid_loop-0.11.1.dist-info/WHEEL +4 -0
  115. froid_loop-0.11.1.dist-info/entry_points.txt +2 -0
  116. froid_loop-0.11.1.dist-info/licenses/LICENSE +30 -0
froid_loop/tui/app.py ADDED
@@ -0,0 +1,1584 @@
1
+ """`froid-loop tui` application shell.
2
+
3
+ Observer/launcher only: the TUI never runs engines in-process. Run control
4
+ (r/s/e) launches detached froid-loop processes in the control session via
5
+ tui.launch (froid-loop-ctl on tmux; a per-registry name on psmux, which the
6
+ launch toasts print). Dry runs are captured into a text modal; validate
7
+ renders its `--json` document into a findings modal (falling back to the text
8
+ one), so the verdict is the document's `ok` rather than an exit code.
9
+ The g binding opens the policy.toml settings editor.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import shlex
15
+ import subprocess
16
+ import time
17
+ from collections.abc import Callable
18
+ from pathlib import Path
19
+ from typing import Any, TypeVar
20
+
21
+ from rich.text import Text
22
+ from textual import work
23
+ from textual.app import App, SuspendNotSupported
24
+ from textual.binding import Binding
25
+ from tomlkit.exceptions import ParseError
26
+
27
+ from .. import froidconfig, decisions, devcontract, policy, resolve, runs, stories, verify
28
+ from ..adapters.multiplexer import MultiplexerError, mux_usable
29
+ from ..journal import load_state
30
+ from ..model import (
31
+ PAUSE_EPIC_BOUNDARY,
32
+ PAUSE_ESCALATION,
33
+ PAUSE_PLAN_CHECKPOINT,
34
+ PAUSE_SPEC_APPROVAL,
35
+ PAUSE_STORY_CHECKPOINT,
36
+ PAUSE_STORY_GATE,
37
+ RunState,
38
+ StoryTask,
39
+ )
40
+ from ..platform_util import resolve_or_lexical
41
+ from ..policy import POLICY_FILE
42
+ from ..process_host import ProcessHostError
43
+ from ..runs import RUNS_DIR, RearmError, StopRunError
44
+ from . import data, launch, widgets
45
+ from .screens.dashboard import DashboardScreen
46
+ from .screens.modals import (
47
+ ConfirmModal,
48
+ ConfirmResumeModal,
49
+ DecisionModal,
50
+ EscalationModal,
51
+ PauseReasonModal,
52
+ SpecReviewModal,
53
+ StartRunModal,
54
+ StartSweepModal,
55
+ StoryCheckpointModal,
56
+ TextOutputModal,
57
+ ValidateFindingsModal,
58
+ )
59
+ from .screens.settings_screen import SettingsScreen
60
+ from .settings import PolicyDoc
61
+
62
+
63
+ def _engine_possibly_live(run_dir: Path) -> bool:
64
+ live = data.liveness(run_dir)
65
+ if live == "alive": # provably live, pid-backed or via a legacy session
66
+ return True
67
+ # 'unknown' means possibly-live only for a pid-backed run (a win32 engine
68
+ # whose pid exists but is unreadable). A legacy pid-less run's 'unknown' just
69
+ # means no session was found — it must not flag every old finished run.
70
+ return live == "unknown" and runs.read_pid(run_dir) is not None
71
+
72
+
73
+ _T = TypeVar("_T")
74
+
75
+
76
+ class FroidLoopApp(App[None]):
77
+ TITLE = "froid-loop"
78
+
79
+ CSS = """
80
+ #left {
81
+ width: 34;
82
+ /* the divider to #detail is the draggable #split-main bar, not a border */
83
+ }
84
+ #runs {
85
+ height: 2fr;
86
+ min-height: 4;
87
+ border-top: solid $primary-darken-2;
88
+ }
89
+ #runs {
90
+ border-title-color: $text;
91
+ border-title-style: bold;
92
+ }
93
+ #sprint-tree, #stories-table {
94
+ /* the dividers above these panes are the draggable splitter bars, which
95
+ also carry the section title that used to ride the border-top */
96
+ height: 3fr;
97
+ min-height: 4;
98
+ }
99
+ #deferred {
100
+ height: 2fr;
101
+ min-height: 4;
102
+ /* strip OptionList's default tall border + padding so the pane sits
103
+ flush with the splitter bar above it */
104
+ border: none;
105
+ padding: 0;
106
+ text-wrap: nowrap;
107
+ text-overflow: ellipsis;
108
+ }
109
+ #detail {
110
+ width: 1fr;
111
+ }
112
+ #runheader {
113
+ height: auto;
114
+ padding: 0 1;
115
+ background: $boost;
116
+ border-bottom: solid $primary-darken-2;
117
+ }
118
+ #tasks {
119
+ height: auto;
120
+ max-height: 35%;
121
+ }
122
+ #tabs {
123
+ height: 1fr;
124
+ }
125
+ #journal {
126
+ height: 1fr;
127
+ }
128
+ """
129
+
130
+ BINDINGS = [
131
+ Binding("q", "quit", "quit"),
132
+ Binding("r", "start_run", "run"),
133
+ Binding("s", "start_sweep", "sweep"),
134
+ Binding("e", "resume_run", "resume"),
135
+ Binding("p", "review_pause", "review"),
136
+ Binding("R", "resolve_run", "resolve"),
137
+ Binding("d", "answer_decisions", "decisions"),
138
+ Binding("a", "attach", "attach"),
139
+ Binding("x", "stop_run", "stop"),
140
+ Binding("S", "graceful_stop_run", "soft-stop"),
141
+ Binding("D", "delete_run", "delete"),
142
+ Binding("A", "archive_run", "archive"),
143
+ Binding("c", "cleanup_sessions", "cleanup"),
144
+ Binding("v", "validate", "validate"),
145
+ Binding("g", "settings", "settings"),
146
+ Binding("M", "toggle_dark", "mode"),
147
+ ]
148
+
149
+ def __init__(self, project: Path):
150
+ super().__init__()
151
+ self.project = resolve_or_lexical(project)
152
+ self.sub_title = str(self.project)
153
+ self._dashboard = DashboardScreen(self.project)
154
+
155
+ def on_mount(self) -> None:
156
+ self.push_screen(self._dashboard)
157
+
158
+ def action_toggle_dark(self) -> None:
159
+ self.theme = "textual-light" if self.theme == "textual-dark" else "textual-dark"
160
+
161
+ # ------------------------------------------------------------ run control
162
+
163
+ def _mux_missing(self) -> bool:
164
+ if launch.mux_available():
165
+ return False
166
+ self.notify("multiplexer backend unavailable — launch/attach disabled", severity="error")
167
+ return True
168
+
169
+ def _mux_guarded(self, probe: Callable[[], _T]) -> tuple[bool, _T | None]:
170
+ """Run a raiser-side multiplexer *read* probe from a foreground action
171
+ handler, converting a transport failure into an error toast. Returns
172
+ (ok, value); when ok is False a MultiplexerError was caught and toasted
173
+ and the handler must abort — a backend hiccup after the availability
174
+ pre-gate fails the action soft instead of crashing the TUI. Foreground
175
+ only: worker threads marshal notify() via call_from_thread (see
176
+ _cleanup_sessions_worker), and launch-layer failures convert to
177
+ LaunchError (see launch._ensure_ctl_session)."""
178
+ try:
179
+ return True, probe()
180
+ except MultiplexerError as e:
181
+ self.notify(str(e), severity="error")
182
+ return False, None
183
+
184
+ def _guarded(self, go: Callable[[], None]) -> None:
185
+ """Pre-launch guard mirroring the CLI: the git support floor refused first,
186
+ then the #414 isolation/repo_root conflict, then a clean worktree required,
187
+ plus a confirm when another engine is already live."""
188
+ # First, in `cmd_run`'s own order, and for the reason that order exists: this
189
+ # is a fact about the HOST, so every other answer here would be advice about
190
+ # the wrong thing. Without it the operator got the generic "launch may have
191
+ # failed — attach to control session" toast the dashboard raises 10s later,
192
+ # which names neither git nor the floor and points at a pane to go read.
193
+ #
194
+ # `timeout_s` because this runs on the event loop — the same reason the
195
+ # commit-subject probe below carries one (`_commit_subject`) — and
196
+ # that bound is exactly why a `GitError` here FALLS THROUGH to launch instead
197
+ # of refusing: 5s is not the deadline the detached CLI applies, so a merely
198
+ # slow git would otherwise be refused by a toast on a host the CLI would run
199
+ # on. The guard never authorizes anything — `_reject_under_floor_git` fails
200
+ # closed on that same fault a moment later, in the process that matters —
201
+ # so declining to PRE-EMPT a refusal costs nothing but a slower message.
202
+ # Same disposition, same reasoning, as the unreadable-policy fall-through
203
+ # below: the guard cannot tell "fine" from "could not look".
204
+ try:
205
+ found = verify.git_below_floor(self.project, timeout_s=5)
206
+ except verify.GitError:
207
+ found = None
208
+ if found is not None:
209
+ self.notify(verify.under_floor_git_message(found), severity="error")
210
+ return
211
+ # The detached CLI refuses this combination too, and it is the authority —
212
+ # this only turns a pane that dies immediately into a toast. Ordered ahead of
213
+ # the clean-tree gate for the same reason `cmd_run` orders it ahead: this one
214
+ # says the configuration cannot run at all, so answering "commit or stash
215
+ # first" would send the operator to fix something that is not the problem.
216
+ #
217
+ # An unreadable config or policy falls through to launch rather than
218
+ # blocking: the guard cannot tell "no conflict" from "could not look", so it
219
+ # defers to the CLI, which reads the same two files and fails loudly on
220
+ # whichever one it cannot parse. Both loaders convert an undecodable file
221
+ # into their own typed error, so the two named here are the whole surface;
222
+ # a raw `UnicodeDecodeError` would be a ValueError and escape.
223
+ try:
224
+ conflict = froidconfig.worktree_isolation_conflict(
225
+ froidconfig.load_paths(self.project),
226
+ policy.load(self.project / POLICY_FILE).scm.isolation,
227
+ )
228
+ except (froidconfig.FroidConfigError, policy.PolicyError, OSError):
229
+ conflict = None
230
+ if conflict is not None:
231
+ self.notify(conflict, severity="error")
232
+ return
233
+ try:
234
+ if not verify.worktree_clean(self.project):
235
+ self.notify(
236
+ "git worktree is not clean — commit or stash first",
237
+ severity="error",
238
+ )
239
+ return
240
+ except verify.GitError as e:
241
+ self.notify(f"git check failed: {e}", severity="error")
242
+ return
243
+ live = [
244
+ r.run_id for r in data.discover_runs(self.project) if _engine_possibly_live(r.run_dir)
245
+ ]
246
+ if live:
247
+ self.push_screen(
248
+ ConfirmModal(
249
+ "another run may be live",
250
+ f"live or unknown: {', '.join(live)}\n"
251
+ "launching another engine on the same project may conflict.",
252
+ confirm_label="launch anyway",
253
+ ),
254
+ lambda ok: go() if ok else None,
255
+ )
256
+ else:
257
+ go()
258
+
259
+ def action_start_run(self) -> None:
260
+ if self._mux_missing():
261
+ return
262
+ source, spec_folder = self._stories_defaults()
263
+ self.push_screen(
264
+ StartRunModal(self.project, default_source=source, default_spec_folder=spec_folder),
265
+ self._start_run_result,
266
+ )
267
+
268
+ def _stories_defaults(self) -> tuple[str, str]:
269
+ """The [stories] policy source + spec_folder to prefill the start-run
270
+ modal, or the sprint-mode default when policy is unreadable — including
271
+ undecodable, which `policy.load` reports as a `PolicyError` rather than
272
+ letting a raw `UnicodeDecodeError` past this handler.
273
+
274
+ `ParseError` is not in the tuple: `policy.load` parses with `tomllib`, so
275
+ the tomlkit error can only arrive from the `PolicyDoc` path in
276
+ :meth:`action_settings`, which catches it there."""
277
+ try:
278
+ pol = policy.load(self.project / POLICY_FILE)
279
+ except (policy.PolicyError, OSError):
280
+ return "sprint-status", ""
281
+ return pol.stories.source, pol.stories.spec_folder
282
+
283
+ def _start_run_result(self, result: dict | None) -> None:
284
+ if not result:
285
+ return
286
+ stories_on = result["source"] == "stories"
287
+ spec_folder = result["spec_folder"] if stories_on else ""
288
+ if stories_on and not spec_folder:
289
+ self.notify("stories mode needs a spec folder", severity="error")
290
+ return
291
+ if result["dry_run"]:
292
+ tail = ["run", "--project", str(self.project), "--dry-run"]
293
+ if stories_on:
294
+ tail += ["--spec", spec_folder]
295
+ if result["epic"] is not None:
296
+ tail += ["--epic", str(result["epic"])]
297
+ if result["story"]:
298
+ tail += ["--story", result["story"]]
299
+ if result["max_stories"] is not None:
300
+ tail += ["--max-stories", str(result["max_stories"])]
301
+ self._show_captured("run --dry-run", tail)
302
+ return
303
+
304
+ def go() -> None:
305
+ run_id = runs.new_run_id()
306
+ try:
307
+ launch.start_run_detached(
308
+ self.project,
309
+ run_id,
310
+ spec=spec_folder or None,
311
+ epic=result["epic"],
312
+ story=result["story"],
313
+ max_stories=result["max_stories"],
314
+ )
315
+ except launch.LaunchError as e:
316
+ self.notify(str(e), severity="error")
317
+ return
318
+ self.notify(
319
+ f"run {run_id} launched (control session {launch.ctl_session(self.project)})"
320
+ )
321
+ self._dashboard.expect_run(run_id)
322
+
323
+ self._guarded(go)
324
+
325
+ def action_start_sweep(self) -> None:
326
+ if self._mux_missing():
327
+ return
328
+ self.push_screen(StartSweepModal(), self._start_sweep_result)
329
+
330
+ def _start_sweep_result(self, result: dict | None) -> None:
331
+ if not result:
332
+ return
333
+ if result["dry_run"]:
334
+ self._show_captured(
335
+ "sweep --dry-run",
336
+ ["sweep", "--project", str(self.project), "--dry-run"],
337
+ )
338
+ return
339
+
340
+ def go() -> None:
341
+ run_id = runs.new_run_id()
342
+ try:
343
+ launch.start_sweep_detached(
344
+ self.project,
345
+ run_id,
346
+ no_prompt=result["no_prompt"],
347
+ decisions_only=result["decisions_only"],
348
+ max_bundles=result["max_bundles"],
349
+ )
350
+ except launch.LaunchError as e:
351
+ self.notify(str(e), severity="error")
352
+ return
353
+ self.notify(
354
+ f"sweep {run_id} launched (control session {launch.ctl_session(self.project)})"
355
+ )
356
+ self._dashboard.expect_run(run_id)
357
+
358
+ self._guarded(go)
359
+
360
+ def action_answer_decisions(self) -> None:
361
+ """Walk the deferred-work decisions past sweeps left unanswered, one
362
+ modal at a time. Each answer is recorded so the next sweep acts on it
363
+ (build -> bundle, close -> closed, keep-open -> recorded) without asking
364
+ again. No tmux/engine needed — this only edits the ledger and store."""
365
+ pending = data.pending_missed_decisions(self.project)
366
+ if not pending:
367
+ self.notify("no unanswered decisions from past sweeps")
368
+ return
369
+ self._walk_decisions(list(pending), 0, 0)
370
+
371
+ def _walk_decisions(self, pending: list, idx: int, answered: int) -> None:
372
+ if idx >= len(pending):
373
+ if answered:
374
+ self.notify(f"recorded {answered} decision(s) — run a sweep to act on any builds")
375
+ self._dashboard._tick(force_rescan=True)
376
+ return
377
+ decision = pending[idx]
378
+
379
+ def on_choice(option: object | None) -> None:
380
+ if option is None: # skipped this one: stop, keep the rest pending
381
+ if answered:
382
+ self.notify(f"recorded {answered} decision(s)")
383
+ self._dashboard._tick(force_rescan=True)
384
+ return
385
+ ok = self._record_decision(decision, option)
386
+ self._walk_decisions(pending, idx + 1, answered + (1 if ok else 0))
387
+
388
+ self.push_screen(DecisionModal(decision), on_choice)
389
+
390
+ def _record_decision(self, decision: object, option: object) -> bool:
391
+ # decision/option cross the widget boundary as `object`; their runtime types
392
+ # are the Decision/DecisionOption that apply_pre_answer and `.id` expect.
393
+ try:
394
+ decisions.apply_pre_answer(
395
+ self.project,
396
+ decision, # pyright: ignore[reportArgumentType]
397
+ option, # pyright: ignore[reportArgumentType]
398
+ date=time.strftime("%Y-%m-%d"),
399
+ )
400
+ except (OSError, froidconfig.FroidConfigError, ValueError, runs.StateRootError) as e:
401
+ # ValueError is the ledger writers' date precondition; it cannot fire
402
+ # from the strftime above. StateRootError is reachable: the ledger
403
+ # write now takes a cross-process lock whose sidecar lives under the
404
+ # state root (#286/#469), and an environment that names no usable root
405
+ # raises it — it is NOT an OSError, so the tuple has to say so.
406
+ # OSError covers the acquisition itself failing against a live holder.
407
+ # Every one of them uncaught here escapes into the Textual event loop
408
+ # and takes the dashboard down mid-walk — the per-decision
409
+ # notification is the right degradation for a modal the human is still
410
+ # stepping through, and the walk continues to the next decision.
411
+ self.notify(
412
+ f"failed to record {decision.id}: {e}", # pyright: ignore[reportAttributeAccessIssue]
413
+ severity="error",
414
+ )
415
+ return False
416
+ return True
417
+
418
+ def action_resume_run(self) -> None:
419
+ if self._mux_missing():
420
+ return
421
+ run_id = self._dashboard.selected_run_id
422
+ if run_id is None:
423
+ self.notify("no run selected", severity="warning")
424
+ return
425
+ run_dir = self.project / RUNS_DIR / run_id
426
+ try:
427
+ state = load_state(run_dir)
428
+ except (OSError, KeyError, ValueError):
429
+ self.notify(f"state for run {run_id} is unreadable", severity="error")
430
+ return
431
+ if state.finished:
432
+ self.notify(f"run {run_id} already finished", severity="warning")
433
+ return
434
+ engine_alive = _engine_possibly_live(run_dir)
435
+
436
+ def done(ok: bool | None) -> None:
437
+ # Re-check liveness at confirm time via the shared guard — the modal's
438
+ # warning is display-only and sampled at open time, so route through
439
+ # _do_resume (like the e/viewer and re-arm paths) rather than launching
440
+ # blind; a newly-live engine is caught even if the user confirmed.
441
+ if ok:
442
+ self._do_resume(run_id)
443
+
444
+ self.push_screen(ConfirmResumeModal(run_id, state, engine_alive), done)
445
+
446
+ def action_attach(self) -> None:
447
+ if self._mux_missing():
448
+ return
449
+ run_id = self._dashboard.selected_run_id
450
+ if run_id is None:
451
+ self.notify("no run selected", severity="warning")
452
+ return
453
+ session = runs.session_name(run_id)
454
+ win_id = launch.ctl_window_id(self.project, run_id)
455
+ ok, agent_live = self._mux_guarded(lambda: launch.session_exists(session))
456
+ if not ok:
457
+ return
458
+ # A sweep blocked on a decision prompt has no agent session — the
459
+ # human answers in the orchestrator's ctl window. Otherwise prefer the
460
+ # live agent session, falling back to the ctl window between sessions.
461
+ if win_id is not None and (self._dashboard.decision_pending is not None or not agent_live):
462
+ launch.select_ctl_window_id(win_id)
463
+ self._attach_to_target(launch.ctl_target(self.project), return_window=win_id)
464
+ return
465
+ elif agent_live:
466
+ target = runs.session_target(run_id)
467
+ else:
468
+ self.notify(
469
+ f"nothing to attach: no live agent session ({session}) and no "
470
+ f"{launch.ctl_session(self.project)} window for this run (runs started outside "
471
+ "the TUI have none)",
472
+ severity="warning",
473
+ timeout=10,
474
+ )
475
+ return
476
+ self._attach_to_target(target)
477
+
478
+ def _attach_to_target(self, target: str, return_window: str | None = None) -> None:
479
+ ok, argv = self._mux_guarded(lambda: runs.attach_target_argv(target))
480
+ if not ok:
481
+ return
482
+ # argv is None only when `ok` is False (they are a pair); past this guard it
483
+ # is a real argv, but pyright can't correlate the two — hence the
484
+ # reportArgumentType ignores on the subprocess.call/shlex.join uses below.
485
+ # Backend-honest inside-the-multiplexer probe (current_return_target()
486
+ # is None outside): inside, attach_target_argv returned the
487
+ # fire-and-forget switch/focus form, so no suspend is needed.
488
+ ret = launch.current_return_target()
489
+ if ret is not None:
490
+ # Record our own pane target (session-qualified when resolvable)
491
+ # on the ctl window so its trailing shell switches the client back
492
+ # here when it exits, instead of stranding the user in the control
493
+ # session.
494
+ if return_window is not None:
495
+ launch.set_return_pane(return_window, ret)
496
+ subprocess.call(argv) # pyright: ignore[reportArgumentType]
497
+ return
498
+ # Outside tmux we attach a throwaway client (under suspend). The ctl
499
+ # session keeps its own shell window, so a closed run window would leave
500
+ # that client parked on the shell rather than ending the attach; tell the
501
+ # window to detach the client on exit so `tmux attach` returns and the
502
+ # TUI resumes where the user left it.
503
+ if return_window is not None:
504
+ launch.set_return_pane(return_window, launch.RETURN_DETACH)
505
+ try:
506
+ with self.suspend():
507
+ subprocess.call(argv) # pyright: ignore[reportArgumentType]
508
+ except SuspendNotSupported:
509
+ self.notify(
510
+ f"cannot suspend here — run manually: {shlex.join(argv)}", # pyright: ignore[reportArgumentType]
511
+ severity="warning",
512
+ timeout=10,
513
+ )
514
+
515
+ def action_resolve_run(self) -> None:
516
+ if self._mux_missing():
517
+ return
518
+ run_id = self._dashboard.selected_run_id
519
+ if run_id is None:
520
+ self.notify("no run selected", severity="warning")
521
+ return
522
+ run_dir = self.project / RUNS_DIR / run_id
523
+ try:
524
+ state = load_state(run_dir)
525
+ except (OSError, KeyError, ValueError):
526
+ self.notify(f"state for run {run_id} is unreadable", severity="error")
527
+ return
528
+ if state.paused_stage != "escalation":
529
+ self.notify(
530
+ "resolve is only available for a run paused at an escalation",
531
+ severity="warning",
532
+ )
533
+ return
534
+ if _engine_possibly_live(run_dir):
535
+ self.notify(f"run {run_id} may still be live — stop it first", severity="warning")
536
+ return
537
+ story = state.paused_story_key or "?"
538
+
539
+ self.push_screen(
540
+ ConfirmModal(
541
+ "resolve escalation",
542
+ f"open the resolve agent for {story}?\n"
543
+ "converse to fix the frozen spec, then confirm re-arm + resume in that window.",
544
+ confirm_label="resolve",
545
+ ),
546
+ lambda ok: self._launch_resolve(run_id) if ok else None,
547
+ )
548
+
549
+ def _launch_resolve(self, run_id: str) -> None:
550
+ """Open the interactive resolve agent for run_id in a ctl window and
551
+ attach — the same path `froid-loop resolve` drives. The caller has already
552
+ confirmed and (for the escalation viewer) gated on liveness."""
553
+ try:
554
+ win_id = launch.start_resolve_detached(self.project, run_id)
555
+ except launch.LaunchError as e:
556
+ self.notify(str(e), severity="error")
557
+ return
558
+ if not win_id:
559
+ self.notify("resolve launched but its window id was not captured", severity="error")
560
+ return
561
+ if not launch.ctl_window_recorded(self.project, run_id, win_id):
562
+ # Not an error and not a reason to abort: this attach targets the id
563
+ # in hand, so the resolve session itself is reached correctly. What
564
+ # is lost is the record *later* verbs read, so `a`/`x` after this
565
+ # window is minted may answer an older one (#482's symptom).
566
+ self.notify(
567
+ "resolve launched but its window id was not recorded — "
568
+ "later attach/stop may target an older window for this run",
569
+ severity="warning",
570
+ )
571
+ launch.select_ctl_window_id(win_id)
572
+ self._attach_to_target(launch.ctl_target(self.project), return_window=win_id)
573
+
574
+ # -------------------------------------------------------- HITL pause review
575
+
576
+ def action_review_pause(self) -> None:
577
+ """Open the stage-appropriate review viewer for the selected paused run.
578
+ Each viewer's actions call the exact code paths the CLI uses (resume,
579
+ reset-to-draft + resume, rearm + resume, resolve, stop) — no duplicated
580
+ logic. Pause kind is read from RunState.paused_stage."""
581
+ selected = self._paused_selection()
582
+ if selected is None:
583
+ return
584
+ run_id, run_dir, state = selected
585
+ stage = state.paused_stage
586
+ if stage == PAUSE_PLAN_CHECKPOINT:
587
+ self._review_plan_checkpoint(run_id, run_dir, state)
588
+ elif stage == PAUSE_STORY_CHECKPOINT:
589
+ self._review_story_checkpoint(run_id, run_dir, state)
590
+ elif stage == PAUSE_ESCALATION:
591
+ self._review_escalation(run_id, run_dir, state)
592
+ elif stage in (PAUSE_SPEC_APPROVAL, PAUSE_EPIC_BOUNDARY, PAUSE_STORY_GATE):
593
+ self._review_gate(run_id, run_dir, state)
594
+ else:
595
+ self.notify(f"no review viewer for pause stage {stage!r}", severity="warning")
596
+
597
+ def _paused_selection(self) -> tuple[str, Path, RunState] | None:
598
+ run_id = self._dashboard.selected_run_id
599
+ if run_id is None:
600
+ self.notify("no run selected", severity="warning")
601
+ return None
602
+ run_dir = self.project / RUNS_DIR / run_id
603
+ try:
604
+ state = load_state(run_dir)
605
+ except (OSError, KeyError, ValueError):
606
+ self.notify(f"state for run {run_id} is unreadable", severity="error")
607
+ return None
608
+ if not state.paused:
609
+ self.notify("run is not paused — nothing to review", severity="warning")
610
+ return None
611
+ return run_id, run_dir, state
612
+
613
+ def _review_plan_checkpoint(self, run_id: str, run_dir: Path, state: RunState) -> None:
614
+ spec_path, spec_text, readable = self._paused_spec(state)
615
+ modal = SpecReviewModal(
616
+ title="plan checkpoint — review the planned spec before implementation",
617
+ subtitle=self._story_subtitle(state),
618
+ spec_path=spec_path,
619
+ spec_text=spec_text,
620
+ unreadable=not readable,
621
+ actions=[
622
+ ("approve", "Approve & resume", "primary"),
623
+ ("replan", "Request replan", "warning"),
624
+ ],
625
+ )
626
+
627
+ def done(verb: str | None) -> None:
628
+ if verb == "approve":
629
+ self._do_resume(run_id)
630
+ elif verb == "replan":
631
+ if spec_path is None:
632
+ self.notify("no spec file to reset for replan", severity="error")
633
+ return
634
+ self._do_replan(run_id, spec_path, self._paused_spec_root(state))
635
+
636
+ self.push_screen(modal, done)
637
+
638
+ def _review_gate(self, run_id: str, run_dir: Path, state: RunState) -> None:
639
+ label = widgets.pause_label(state.paused_stage or "")[0] or "gate"
640
+ spec_path, spec_text, readable = self._paused_spec(state)
641
+
642
+ def done(verb: str | None) -> None:
643
+ if verb == "resume":
644
+ self._do_resume(run_id)
645
+
646
+ if spec_path is None:
647
+ # Spec-less gates: story-gate fires before the story is registered in
648
+ # state.tasks (deliberate, so a resume re-picks and re-asks the ledger)
649
+ # and epic-boundary has no story key. The pause reason is the payload.
650
+ subtitle = (
651
+ self._story_subtitle(state)
652
+ if state.paused_story_key
653
+ else Text(f"run {run_id}", style="bold")
654
+ )
655
+ self.push_screen(
656
+ PauseReasonModal(
657
+ title=f"{label} — pause reason",
658
+ subtitle=subtitle,
659
+ reason=state.paused_reason or "",
660
+ ),
661
+ done,
662
+ )
663
+ return
664
+ modal = SpecReviewModal(
665
+ title=f"{label} — review the finalized spec",
666
+ subtitle=self._story_subtitle(state),
667
+ spec_path=spec_path,
668
+ spec_text=spec_text,
669
+ unreadable=not readable,
670
+ actions=[("resume", "Approve & resume", "primary")],
671
+ )
672
+ self.push_screen(modal, done)
673
+
674
+ @staticmethod
675
+ def _checkpoint_gate_line(review_cycle: int) -> str:
676
+ """The story-checkpoint card's gate line, derived from real task state.
677
+
678
+ A done_checkpoint fires only after the story's verify + review gates
679
+ passed and it committed, so the pass is backed by the commit's existence
680
+ — but we do not persist per-command verify output, so we state the gates
681
+ cleared plus the follow-up review-cycle count the task actually records,
682
+ never a blanket hardcoded "verification passed" claim."""
683
+ if review_cycle == 0:
684
+ note = "no follow-up review cycles"
685
+ elif review_cycle == 1:
686
+ note = "1 follow-up review cycle"
687
+ else:
688
+ note = f"{review_cycle} follow-up review cycles"
689
+ return f"verify + review gates passed · {note}"
690
+
691
+ def _review_story_checkpoint(self, run_id: str, run_dir: Path, state: RunState) -> None:
692
+ story_key = state.paused_story_key or "?"
693
+ task = state.tasks.get(story_key)
694
+ commit = ""
695
+ tokens = "-"
696
+ # Defensive default: a done_checkpoint implies a commit, but if none is
697
+ # recorded say so rather than assert a verify outcome we cannot back.
698
+ verify_line = "no commit recorded for this story"
699
+ if task is not None:
700
+ if task.commit_sha:
701
+ subject = self._commit_subject(task.commit_sha)
702
+ commit = f"{task.commit_sha[:12]} {subject}".strip()
703
+ verify_line = self._checkpoint_gate_line(task.review_cycle)
704
+ weight = state.cache_read_weight()
705
+ raw = task.tokens.total
706
+ if raw:
707
+ tokens = f"{task.tokens.weighted_total(weight):,} ({raw:,} raw)"
708
+ modal = StoryCheckpointModal(
709
+ story_key=story_key,
710
+ title=self._story_context(state, story_key)[0],
711
+ commit=commit,
712
+ verify_line=verify_line,
713
+ tokens=tokens,
714
+ )
715
+
716
+ def done(verb: str | None) -> None:
717
+ if verb == "continue":
718
+ self._do_resume(run_id)
719
+ elif verb == "stop":
720
+ self._stop_run_worker(run_id, run_dir)
721
+
722
+ self.push_screen(modal, done)
723
+
724
+ def _review_escalation(self, run_id: str, run_dir: Path, state: RunState) -> None:
725
+ story_key = state.paused_story_key or "?"
726
+ spec_path, spec_text, readable = self._paused_spec(state)
727
+ title, description = self._story_context(state, story_key)
728
+ restore_recorded = self._restore_recorded(run_dir, story_key)
729
+ modal = EscalationModal(
730
+ story_key=story_key,
731
+ title=title,
732
+ description=description,
733
+ # `_blocking_condition` reduces the read-failure body to "" like any
734
+ # other text without a halt block, so an unreadable spec would render
735
+ # "(no blocking condition recorded)" — indistinguishable from a spec that
736
+ # was read fine and simply halted without one. The verdict has to be
737
+ # carried in, and it also REFUSES both verbs: re-arm flips the spec's
738
+ # frontmatter, strips its result and re-stamps the baseline, which is not
739
+ # an action to take on evidence nobody could read.
740
+ blocking=self._blocking_condition(spec_text),
741
+ unreadable=not readable,
742
+ sentinel_kind=self._sentinel_kind(state, story_key),
743
+ resolution_ready=resolve.resolution_path(run_dir, story_key).is_file(),
744
+ engine_live=_engine_possibly_live(run_dir),
745
+ restore_recorded=restore_recorded,
746
+ )
747
+
748
+ def done(verb: str | None) -> None:
749
+ if verb == "resolve":
750
+ if self._mux_missing() or self._resolve_blocked_by_liveness(run_id, run_dir):
751
+ return
752
+ self._launch_resolve(run_id)
753
+ elif verb == "rearm":
754
+ self._do_rearm(run_id, run_dir, story_key, restore_recorded=restore_recorded)
755
+
756
+ self.push_screen(modal, done)
757
+
758
+ @staticmethod
759
+ def _restore_recorded(run_dir: Path, story_key: str) -> bool:
760
+ """True when resolution.json records — or, being unreadable, MAY record —
761
+ a restore_patch. The TUI re-arm path is a plain from-scratch re-drive
762
+ (only the CLI resolve flow honors the latch, because a stale marker is
763
+ indistinguishable from a fresh one here), so a recorded restore must be
764
+ surfaced rather than silently dropped."""
765
+ if not resolve.resolution_path(run_dir, story_key).is_file():
766
+ return False
767
+ try:
768
+ doc = resolve.read_resolution(run_dir, story_key)
769
+ except resolve.ResolutionError:
770
+ return True # can't prove it carries no restore — surface the warning
771
+ return bool(doc and doc.get("restore_patch"))
772
+
773
+ # --------------------------------------------------- shared pause code paths
774
+
775
+ def _do_resume(self, run_id: str) -> None:
776
+ """Resume a paused run — the `froid-loop resume` / `e` path, minus the
777
+ confirm modal (the viewer was the confirmation). Guards tmux + a
778
+ possibly-live engine so an approve/continue can't double-drive. No
779
+ control-alias gate here: this path mutates nothing before the launch,
780
+ and the launcher itself refuses at the mutation's chokepoint
781
+ (`launch.start_detached`) — the LaunchError lands in the except below."""
782
+ if self._mux_missing():
783
+ return
784
+ run_dir = self.project / RUNS_DIR / run_id
785
+ if _engine_possibly_live(run_dir):
786
+ self.notify(f"run {run_id} may still be live — stop it first", severity="warning")
787
+ return
788
+ try:
789
+ win_id = launch.resume_detached(self.project, run_id)
790
+ except launch.LaunchError as e:
791
+ self.notify(str(e), severity="error")
792
+ return
793
+ if not win_id:
794
+ # The resume itself is running; only the disambiguation record is
795
+ # lost, so `a`/`x` may target an older same-run_id window (#482's
796
+ # symptom). Warn instead of masking it behind the success toast.
797
+ # "not recorded", not "not captured": resume_detached reports the
798
+ # uncaptured id and the unwritten record through this one signal
799
+ # because they leave the operator in the same place.
800
+ self.notify(
801
+ "resume launched but its window id was not recorded — "
802
+ "attach/stop may target an older window for this run",
803
+ severity="warning",
804
+ )
805
+ self.notify(
806
+ f"resume of {run_id} launched (control session {launch.ctl_session(self.project)})"
807
+ )
808
+
809
+ def _do_replan(self, run_id: str, spec_path: Path, confine_root: Path) -> None:
810
+ """Request-replan: reset the planned spec to draft + strip its Auto Run
811
+ Result, then resume — the next dispatch re-enters step-02 planning. Uses
812
+ the same devcontract primitives the engine's repair path uses.
813
+
814
+ `confine_root` arrives from the caller (`_paused_spec_root`) rather than being
815
+ `self.project` here: this method has no task in scope, and the root these two
816
+ writers validate against must be the SAME claim about which tree owns the spec
817
+ that `_paused_spec` anchored the path on. `runs.task_spec_root`'s docstring
818
+ carries the rationale — a `confine_root` that disagrees with the anchor is not
819
+ REFUSED, it silently drops both writes to the plain no-follow arm and loses the
820
+ confined arm's O_NOFOLLOW walk (#593) with no signal at all."""
821
+ # Guard a possibly-live engine BEFORE mutating the spec — a draft-reset +
822
+ # strip under a still-running session would race its writes (the rearm path
823
+ # already checks liveness first; match it so replan can't corrupt a live
824
+ # drive, and only then does _do_resume re-check before relaunching).
825
+ # The control-alias gate sits equally early: the child `froid-loop resume`
826
+ # would refuse such a run anyway, and a spec rewritten ahead of that
827
+ # refusal is the mutate-then-refuse shape the CLI entry gates closed.
828
+ if self._blocked_by_control_alias(run_id):
829
+ return
830
+ run_dir = self.project / RUNS_DIR / run_id
831
+ if self._resolve_blocked_by_liveness(run_id, run_dir):
832
+ return
833
+ if not spec_path.is_file():
834
+ # `reset_spec_status` returns False for an ABSENT spec and for one with no
835
+ # frontmatter status alike, and the shared notice below blamed the
836
+ # frontmatter for both. Now that the path is re-anchored on the run's own
837
+ # tree, an absent spec is the signal that the ANCHORING is wrong, so it
838
+ # earns its own message naming the path actually consulted.
839
+ self.notify(f"replan: no spec at {spec_path} — not resuming", severity="error")
840
+ return
841
+ try:
842
+ reset = devcontract.reset_spec_status(spec_path, "draft", confine_root=confine_root)
843
+ devcontract.strip_auto_run_result(spec_path, confine_root=confine_root)
844
+ except (OSError, UnicodeDecodeError, verify.FrontmatterWriteError) as e:
845
+ # FrontmatterWriteError is not an OSError: a spec whose `status:` is a
846
+ # block scalar or a flow mapping reads fine and fails the WRITE. It
847
+ # lands in the same notice as a permissions failure because it has the
848
+ # same shape for the operator — the replan did not happen and the run
849
+ # is not resumed — and because an uncaught raise inside a Textual
850
+ # worker takes the dashboard down instead of saying so.
851
+ #
852
+ # UnicodeDecodeError is a ValueError, so neither sibling arm caught it and
853
+ # `reset_spec_status` decodes STRICTLY (`read_bytes().decode("utf-8")`).
854
+ # That raise became reachable when `_paused_spec` started degrading a
855
+ # non-UTF-8 spec in place instead of raising at render: the operator can now
856
+ # open the modal on one and press replan, which is precisely the event-loop
857
+ # crash the read-side fix exists to prevent.
858
+ self.notify(f"replan failed: {e}", severity="error")
859
+ return
860
+ if not reset:
861
+ # honor the reset bool: nothing was flipped (the spec has no frontmatter
862
+ # status, or is already draft), so the next dispatch would NOT re-enter
863
+ # planning. Surface it instead of a misleading "reset" notice + resume.
864
+ self.notify(
865
+ "replan: could not reset the plan to draft (no frontmatter status?) — not resuming",
866
+ severity="error",
867
+ )
868
+ return
869
+ self.notify("plan reset to draft — the next dispatch re-plans")
870
+ self._do_resume(run_id)
871
+
872
+ def _echo_rearm_events(self, run_dir: Path, before: list[dict[str, Any]] | None) -> bool:
873
+ """Toast the re-arm records `cli._echo_rearm_events` prints, same table.
874
+
875
+ Reads through `runs.journal_entries_or_none`, shared with the CLI so the two
876
+ surfaces cannot drift on robustness the way they drifted on routing. Both ends
877
+ of the diff must be readable: a failed FIRST read degraded to `[]` would set the
878
+ watermark to zero and replay every historical record as a fresh toast, so an
879
+ unreadable journal costs the echo and keeps the gesture.
880
+
881
+ The table's `next_step` is deliberately dropped: it reads "... before
882
+ resuming", and this path resumes in the same gesture.
883
+
884
+ Returns True when a record HOLDS that gesture (`runs.rearm_holds_the_resume`),
885
+ which is the one case where the dropped imperative was load-bearing rather than
886
+ moot — `_do_rearm` stops instead of resuming, and says so in its own words.
887
+ """
888
+ after = runs.journal_entries_or_none(run_dir)
889
+ if before is None or after is None:
890
+ return False
891
+ holds = False
892
+ for entry in after[len(before) :]:
893
+ # before the routing table can drop it: a `None` notice means "nothing to
894
+ # toast", never "nothing to decide"
895
+ holds = runs.rearm_holds_the_resume(entry) or holds
896
+ notice = runs.rearm_event_notice(entry)
897
+ if notice is None:
898
+ continue
899
+ severity, message, _next_step = notice
900
+ self.notify(message, severity="warning" if severity == "warning" else "information")
901
+ return holds
902
+
903
+ def _do_rearm(
904
+ self, run_id: str, run_dir: Path, story_key: str, *, restore_recorded: bool = False
905
+ ) -> None:
906
+ """Re-arm a resolved escalation + resume — the `resolve --no-interactive`
907
+ path (rearm_escalation handles sentinel auto-delete-with-preservation)."""
908
+ # Ahead of rearm_escalation for the same reason cmd_resolve gates at
909
+ # entry: a run left re-armed-but-not-running by the child's refusal.
910
+ if self._blocked_by_control_alias(run_id):
911
+ return
912
+ if self._resolve_blocked_by_liveness(run_id, run_dir):
913
+ return
914
+ # The LIVE isolation mode, read once and used twice below. `runs.rearm_escalation`
915
+ # requires it: how the re-drive WILL run is a policy question, and the recorded
916
+ # `task.worktree_path` answers only how the escalated attempt ran — the two part
917
+ # company on exactly the mid-run policy edit the conflict check below is also
918
+ # about.
919
+ #
920
+ # Unreadable REFUSES here, unlike the launch guard above and unlike this block's
921
+ # own previous disposition. That fall-through was correct while the policy fed
922
+ # one optional CHECK: "no conflict" and "could not look" are different answers
923
+ # and neither blocks a launch the detached CLI will re-read the same file for.
924
+ # It is not correct for an INPUT to a repair write. Without the mode this
925
+ # gesture cannot say which ref the re-drive reads, so it would flip the spec and
926
+ # then tell the operator to put the correction in a tree picked by a default —
927
+ # silently, and unrecoverably, since a re-arm consumes the escalation.
928
+ # `cli.cmd_resolve` raises on the same unreadable file before it re-arms.
929
+ try:
930
+ isolation = policy.load(self.project / POLICY_FILE).scm.isolation
931
+ except (policy.PolicyError, OSError) as e:
932
+ self.notify(
933
+ f"cannot read policy.toml to determine the re-drive's isolation mode "
934
+ f"({e}) — fix it, then re-arm; the story is still escalated",
935
+ severity="error",
936
+ )
937
+ return
938
+ # Same seam as `cli.cmd_resolve`, for the same reason and at the same moment:
939
+ # `runs.rearm_escalation` reads the persisted code root back out of the run
940
+ # state, and only a process that has just read config.yaml can tell whether a
941
+ # `repo_root:` edit made while the run was paused has moved it. Resume re-stamps
942
+ # it, but this gesture re-arms BEFORE it resumes, so the mirror has to be aimed
943
+ # here or the re-arm advances the baseline in the tree the run has left.
944
+ try:
945
+ paths = froidconfig.load_paths(self.project)
946
+ except (froidconfig.FroidConfigError, OSError) as e:
947
+ self.notify(
948
+ f"cannot read the project config to confirm the code root ({e}) — "
949
+ "re-arming against the root this run recorded",
950
+ severity="warning",
951
+ )
952
+ else:
953
+ # Same hoist as `cli.cmd_resolve`, for the same reason: this gesture
954
+ # re-arms and THEN resumes, so the isolation refusal the detached CLI makes
955
+ # in `_resume_paused_run` landed after the re-stamp had persisted the
956
+ # unsupported root and `rearm_escalation` had advanced the attempt baseline
957
+ # against it. The operator saw "re-armed <story>" and then a pane that
958
+ # refused, with the story no longer escalated for `resolve` to correct.
959
+ #
960
+ # Reads the mode hoisted above rather than loading policy.toml a second
961
+ # time: two reads of one file in one gesture can disagree under a concurrent
962
+ # edit, and the refusal must be about the same mode the re-arm is told.
963
+ conflict = froidconfig.worktree_isolation_conflict(paths, isolation)
964
+ if conflict is not None:
965
+ self.notify(conflict, severity="error")
966
+ return
967
+ if (moved := runs.restamp_code_root(run_dir, paths.repo_root)) is not None:
968
+ self.notify(moved, severity="warning")
969
+ before_entries = runs.journal_entries_or_none(run_dir)
970
+ hold_resume = False
971
+ try:
972
+ runs.rearm_escalation(run_dir, story_key, isolated_redrive=isolation == "worktree")
973
+ except RearmError as e:
974
+ self.notify(f"re-arm failed: {e}", severity="error")
975
+ return
976
+ finally:
977
+ # In the `finally`, matching `cli.cmd_resolve`. `_stale_restore_residue`
978
+ # journals BEFORE the re-stamp block that raises `RearmError`, so on that
979
+ # path the records were already written and returning early threw them
980
+ # away — including `stale-restore-commits`, the one record whose whole
981
+ # point is that nothing else will tell the human. This surface used to
982
+ # `return` there while the CLI echoed, so the two DID drift on the abort
983
+ # path even after they were unified on routing — and an abort is when the
984
+ # residue matters most: the re-arm half-ran and the operator has to decide
985
+ # what to do with the tree.
986
+ hold_resume = self._echo_rearm_events(run_dir, before_entries)
987
+ if restore_recorded:
988
+ self.notify(
989
+ "recorded restore patch NOT honored — this re-arm re-drives from "
990
+ "scratch (only `froid-loop resolve` applies a restore)",
991
+ severity="warning",
992
+ )
993
+ self.notify(f"re-armed {story_key}")
994
+ if hold_resume:
995
+ # The half of the gesture that still worked is kept: the story IS re-armed
996
+ # and persisted. What stops is the resume this surface folds in behind it,
997
+ # because the warning above proved the re-drive would read a spec it cannot
998
+ # route on and burn the escalation. Worded for a surface that drops
999
+ # `next_step`, and worded as an instruction the operator can finish here —
1000
+ # the run stays paused and resumable from this same screen.
1001
+ self.notify(
1002
+ "not resuming: commit the corrected spec, then resume this run",
1003
+ severity="warning",
1004
+ )
1005
+ return
1006
+ self._do_resume(run_id)
1007
+
1008
+ def _resolve_blocked_by_liveness(self, run_id: str, run_dir: Path) -> bool:
1009
+ if _engine_possibly_live(run_dir):
1010
+ self.notify(f"run {run_id} may still be live — stop it first", severity="warning")
1011
+ return True
1012
+ return False
1013
+
1014
+ def _blocked_by_control_alias(self, run_id: str) -> bool:
1015
+ """Refuse to mutate persisted state for a run whose id aliases a
1016
+ control session (`ctl`, `ctl-<16 hex>` — the CLI's resume/resolve
1017
+ gates, mirrored): the launch it would end in is refused at the
1018
+ mutation chokepoint (`launch.start_detached`), so a spec reset or an
1019
+ escalation re-arm performed FIRST would strand the run in the mutated
1020
+ state. Only the paths that mutate before launching need this —
1021
+ `_do_replan` (spec draft-reset/strip) and `_do_rearm`
1022
+ (rearm_escalation); plain resume/resolve mutate nothing early and are
1023
+ covered by the launcher's own gate."""
1024
+ if runs.run_id_aliases_control_session(run_id):
1025
+ self.notify(
1026
+ f"run {run_id}: its agent session name is the control session's own — "
1027
+ "cannot be driven. Recover its work by hand, then `froid-loop delete "
1028
+ f"{run_id}`",
1029
+ severity="error",
1030
+ )
1031
+ return True
1032
+ return False
1033
+
1034
+ # ---------------------------------------------------- pause-context readers
1035
+
1036
+ def _paused_task(self, state: RunState) -> StoryTask | None:
1037
+ """The paused story's task, or None when nothing is paused.
1038
+
1039
+ One lookup for both `_paused_spec` (which anchors the READ) and
1040
+ `_paused_spec_root` (which supplies the destructive write's `confine_root`).
1041
+ The whole point of routing both through `runs.task_spec_path`/`task_spec_root`
1042
+ is that the anchor and the root must name one tree; two copies of the lookup
1043
+ would let them drift on the very state that decides it."""
1044
+ return state.tasks.get(state.paused_story_key) if state.paused_story_key else None
1045
+
1046
+ def _paused_spec(self, state: RunState) -> tuple[Path | None, str, bool]:
1047
+ """(spec path, spec text, readable) for the paused story, or (None, "", True)
1048
+ when the task has no spec file (e.g. an ambiguous-match escalation).
1049
+
1050
+ `readable` is False only when the spec could not be READ at the anchored path,
1051
+ which is the signal that the anchoring is wrong. It is returned rather than left
1052
+ for the renderer to infer, because the alternative is sniffing the body for the
1053
+ failure sentence — the failure text and a spec that merely opens with the same
1054
+ words are not distinguishable after the fact, and one of them must not disable
1055
+ an operator's approve button.
1056
+
1057
+ The path is re-anchored through `runs.task_spec_path`, never `Path(...)` on the
1058
+ raw value: `model.StoryTask._serialized_worktree_path` persists an isolated
1059
+ unit's spec RELATIVE to the mounted worktree root and `from_dict` reads it back
1060
+ raw, so a bare `Path(task.spec_file)` resolves against the TUI process cwd —
1061
+ where the main checkout carries the very same `_froid-output/specs/...` layout
1062
+ and answers with the WRONG tree's copy of the story spec."""
1063
+ task = self._paused_task(state)
1064
+ if task is None or not task.spec_file:
1065
+ return None, "", True
1066
+ path = runs.task_spec_path(task, state)
1067
+ try:
1068
+ # `errors="replace"` for the same reason `_commit_subject` uses it: a story
1069
+ # spec is agent- or human-authored, so an odd byte is a fact about the file,
1070
+ # not a reason to withhold it. Decoding strictly here cost the reviewer the
1071
+ # WHOLE document at a gate whose only purpose is reading it — and, because
1072
+ # every review surface calls this from the Textual event loop, an escaping
1073
+ # UnicodeDecodeError (a ValueError, so no OSError arm catches it) took the
1074
+ # dashboard down rather than rendering the fault.
1075
+ return path, path.read_bytes().decode("utf-8", errors="replace"), True
1076
+ except OSError as e:
1077
+ # An absent spec at the ANCHORED path is the signal that the anchoring is
1078
+ # wrong, so it must not reduce to "" — SpecReviewModal renders that as
1079
+ # "(empty spec)", which is also what a present-but-blank spec renders as.
1080
+ # Report the failure as the body so the two cases read differently, and
1081
+ # keep this arm to ABSENCE now that a decode fault degrades in place.
1082
+ return path, f"(spec could not be read — {e})", False
1083
+
1084
+ def _paused_spec_root(self, state: RunState) -> Path:
1085
+ """The tree the paused story's spec is anchored on — and confined to.
1086
+
1087
+ The mirror of `_paused_spec`'s anchor, kept as a sibling so the three read-only
1088
+ consumers keep the untouched two-value read. `_do_replan` WRITES the path
1089
+ `_paused_spec` returned, and `runs.task_spec_root` is the single definition
1090
+ backing both halves: an anchor and a `confine_root` that name different trees do
1091
+ not refuse, they silently degrade the write (#593).
1092
+
1093
+ The no-task arm is `Path(state.project)`, NOT `self.project`, so both arms make
1094
+ one claim: the delegate answers from the state the run persisted at launch,
1095
+ while `self.project` is the constructor's `resolve_or_lexical` of the operator's
1096
+ argument, and the two can differ. That arm is currently unreachable from the
1097
+ write path — `_review_plan_checkpoint`'s `done()` refuses a `None` `spec_path`
1098
+ before calling `_do_replan`, and `_paused_spec` returns `None` on BOTH of its
1099
+ arms (no task, and a task carrying no `spec_file`) — so this is about not
1100
+ leaving a second claim lying around for a future caller, not a live bug. The
1101
+ no-task arm is the only one reachable here: a task with an empty `spec_file`
1102
+ still answers from `task_spec_root`, which needs no spec to name a tree."""
1103
+ task = self._paused_task(state)
1104
+ return runs.task_spec_root(task, state) if task else Path(state.project)
1105
+
1106
+ def _story_subtitle(self, state: RunState) -> Text:
1107
+ key = state.paused_story_key or "?"
1108
+ title = self._story_context(state, key)[0]
1109
+ text = Text(key, style="bold")
1110
+ if title:
1111
+ text.append(f" — {title}")
1112
+ return text
1113
+
1114
+ def _story_context(self, state: RunState, key: str) -> tuple[str, str]:
1115
+ """(title, description) from stories.yaml in stories mode, else ("", "")."""
1116
+ if state.source != "stories" or not state.spec_folder:
1117
+ return "", ""
1118
+ # `task_stories_root`, not `self.project`, for the reason `_sentinel_kind`
1119
+ # states below: BOTH feed one `EscalationModal` — this supplies its title and
1120
+ # description, that its sentinel indicator — so a manifest read from the main
1121
+ # checkout beside a sentinel read from the mount is the same one-surface-two-trees
1122
+ # defect the anchor exists to close. `self.project` is also the wrong VALUE for
1123
+ # the no-task arm: it is the constructor's `resolve_or_lexical` of the operator's
1124
+ # argument, while every other anchored read here answers from `state.project`,
1125
+ # the path the run persisted at launch.
1126
+ root = runs.task_stories_root(state.tasks.get(key), state)
1127
+ try:
1128
+ folder = stories.resolve_spec_folder(root, state.spec_folder)
1129
+ entry = stories.load_stories(folder).get(key)
1130
+ except stories.StoriesError:
1131
+ return "", ""
1132
+ return (entry.title, entry.description) if entry else ("", "")
1133
+
1134
+ def _sentinel_kind(self, state: RunState, key: str) -> str:
1135
+ if state.source != "stories" or not state.spec_folder:
1136
+ return ""
1137
+ # Anchored on the tree the RUN owns, for the same reason `_paused_spec` is: the
1138
+ # sentinel the engine wrote lives in the unit's mount under isolation
1139
+ # (`stories_engine._stories_folder` IS the worktree during a driven story),
1140
+ # while the main checkout carries the same layout and holds a stale twin or
1141
+ # nothing. Both values feed ONE `EscalationModal` — the spec text through
1142
+ # `_blocking_condition`, this through `sentinel_kind` — so anchoring them on
1143
+ # different trees let a single modal disagree with itself and rendered a
1144
+ # pre-planning sentinel wedge as an ordinary escalation.
1145
+ #
1146
+ # `task_stories_root`, not `task_spec_root`: the folder is located from the
1147
+ # workspace root, and the latter's out-of-mount arm answers a confinement
1148
+ # question about `spec_file` that would send this read to the main checkout
1149
+ # while `_stories_folder` stayed on the mount. It also takes `None`, so the
1150
+ # no-task fallback is not re-spelled here.
1151
+ root = runs.task_stories_root(state.tasks.get(key), state)
1152
+ # resolve_story_spec globs + reads frontmatter; a file removed mid-scan (a
1153
+ # re-arm clearing the sentinel while the viewer refreshes) can raise OSError.
1154
+ # Degrade to "" rather than let a race-window read crash the render.
1155
+ try:
1156
+ folder = stories.resolve_spec_folder(root, state.spec_folder)
1157
+ st = stories.resolve_story_spec(folder, key)
1158
+ except OSError:
1159
+ return ""
1160
+ return st.sentinel_kind if st.kind == stories.KIND_SENTINEL else ""
1161
+
1162
+ @staticmethod
1163
+ def _blocking_condition(spec_text: str) -> str:
1164
+ """The `## Auto Run Result` block a blocked spec records its halt in."""
1165
+ idx = spec_text.find("## Auto Run Result")
1166
+ return spec_text[idx:].strip() if idx != -1 else ""
1167
+
1168
+ def _commit_subject(self, sha: str) -> str:
1169
+ # Through the chokepoint (#390): a timeout or failed spawn arrives as
1170
+ # GitError/GitSpawnError instead of the raw subprocess pair, and taking
1171
+ # bytes closes the strict-decode hole — text=True raised
1172
+ # UnicodeDecodeError (a ValueError, caught by neither arm of the old
1173
+ # guard) on a subject undecodable in the run's codec, crashing the
1174
+ # checkpoint modal. Subject bytes are git's logOutputEncoding (UTF-8
1175
+ # unless configured); replace so an odd byte degrades one label,
1176
+ # never raises mid-render.
1177
+ # timeout_s=5 keeps the pre-#390 deadline: this runs on the event loop
1178
+ # (the checkpoint modal's build path), so a stalled git must surface as
1179
+ # a missing subject in seconds, not a 120s frozen UI.
1180
+ try:
1181
+ proc = verify.git_bytes(self.project, "log", "-1", "--format=%s", sha, timeout_s=5)
1182
+ except verify.GitError:
1183
+ return ""
1184
+ if proc.returncode != 0:
1185
+ return ""
1186
+ return proc.stdout.decode("utf-8", errors="replace").strip()
1187
+
1188
+ # ------------------------------------------------------ stop / delete / archive
1189
+
1190
+ def _selected_run_dir(self) -> tuple[str, Path] | None:
1191
+ run_id = self._dashboard.selected_run_id
1192
+ if run_id is None:
1193
+ self.notify("no run selected", severity="warning")
1194
+ return None
1195
+ return run_id, self.project / RUNS_DIR / run_id
1196
+
1197
+ def action_stop_run(self) -> None:
1198
+ if self._mux_missing():
1199
+ return
1200
+ selected = self._selected_run_dir()
1201
+ if selected is None:
1202
+ return
1203
+ run_id, run_dir = selected
1204
+ if not data.liveness(run_dir) == "alive":
1205
+ self.notify(f"run {run_id} is not live", severity="warning")
1206
+ return
1207
+
1208
+ def done(ok: bool | None) -> None:
1209
+ if ok:
1210
+ self._stop_run_worker(run_id, run_dir)
1211
+
1212
+ self.push_screen(
1213
+ ConfirmModal("stop run", f"stop run {run_id}?", confirm_label="stop"), done
1214
+ )
1215
+
1216
+ @work(thread=True, group="lifecycle")
1217
+ def _stop_run_worker(self, run_id: str, run_dir: Path) -> None:
1218
+ try:
1219
+ runs.stop_run(run_dir)
1220
+ launch.kill_ctl_window(self.project, run_id)
1221
+ except (OSError, StopRunError, ProcessHostError) as e:
1222
+ self.call_from_thread(self.notify, f"stop failed: {e}", severity="error")
1223
+ return
1224
+ self.call_from_thread(self.notify, f"run {run_id} stopped")
1225
+
1226
+ def action_graceful_stop_run(self) -> None:
1227
+ """Ask the selected live run to stop *gracefully*: finish the in-flight item
1228
+ (story dev/review/commit, or a sweep bundle through commit), then finalize
1229
+ cleanly and stop — resumable, unlike the hard stop `x` delivers, which
1230
+ abandons the in-flight item.
1231
+
1232
+ Deliberately no `_mux_missing` gate: unlike `x` (which kills the agent
1233
+ window) this touches no multiplexer — the request rides the same control
1234
+ file a hard stop uses, in its graceful mode, read by the engine at item
1235
+ boundaries — so it must work even with the backend down. The liveness gate is also deliberately looser than `x`'s: it rejects
1236
+ only a *provably dead* engine, so an unverifiable (`unknown`) pid — a win32
1237
+ access-denied pid, a psmux backend, a run on another host — still lodges the
1238
+ request, matching `runs.request_graceful_stop`'s `requested-unverifiable`
1239
+ path (the request stands and fires if an engine is in fact running)."""
1240
+ selected = self._selected_run_dir()
1241
+ if selected is None:
1242
+ return
1243
+ run_id, run_dir = selected
1244
+ if data.liveness(run_dir) == "dead":
1245
+ self.notify(f"run {run_id} is not live", severity="warning")
1246
+ return
1247
+
1248
+ def done(ok: bool | None) -> None:
1249
+ if ok:
1250
+ self._graceful_stop_worker(run_id, run_dir)
1251
+
1252
+ self.push_screen(
1253
+ ConfirmModal(
1254
+ "graceful stop",
1255
+ f"stop run {run_id} after the current item finishes?\n"
1256
+ "the in-flight story/bundle completes through commit, then the run "
1257
+ "finalizes and stops (resumable). `x` instead abandons the "
1258
+ "in-flight item.",
1259
+ confirm_label="graceful stop",
1260
+ ),
1261
+ done,
1262
+ )
1263
+
1264
+ @work(thread=True, group="lifecycle")
1265
+ def _graceful_stop_worker(self, run_id: str, run_dir: Path) -> None:
1266
+ # The TUI is an observer: it only writes the control file via the runs
1267
+ # helper (atomic tmp + replace) — it never signals the engine, shells out,
1268
+ # or writes the journal. request_graceful_stop returns a status token to
1269
+ # message on; every UI update from this thread marshals through
1270
+ # call_from_thread (worker threads must not touch widgets directly).
1271
+ try:
1272
+ outcome = runs.request_graceful_stop(run_dir)
1273
+ except runs.GracefulStopError as e:
1274
+ self.call_from_thread(self.notify, str(e), severity="error")
1275
+ return
1276
+ except OSError as e:
1277
+ # Mirrors the CLI's `stop --graceful` arm: the lodge does not roll
1278
+ # back a failed write (see _create_stop_request), so a part-way
1279
+ # failure can leave a graceful request standing — and a confined
1280
+ # refusal (`UnconfinedWriteError`, #593) wrote nothing at all. Either
1281
+ # way the worker must not raise: Textual workers default to
1282
+ # exit_on_error=True, so an uncaught OSError here would take the
1283
+ # whole dashboard down instead of reporting the refusal.
1284
+ self.call_from_thread(
1285
+ self.notify,
1286
+ f"run {run_id}: stop request could not be written ({e}) — a graceful "
1287
+ f"request may still be pending; check `froid-loop status {run_id}` and "
1288
+ f"use `froid-loop stop {run_id} --cancel-graceful` to withdraw it",
1289
+ severity="error",
1290
+ )
1291
+ return
1292
+ if outcome == "already-pending":
1293
+ # Mode-neutral: the pending request may be a hard one, and this token
1294
+ # cannot tell (#319) — same wording as the CLI's `stop --graceful`.
1295
+ self.call_from_thread(self.notify, f"run {run_id} already has a stop request pending")
1296
+ return
1297
+ if outcome == "requested-unverifiable":
1298
+ self.call_from_thread(
1299
+ self.notify,
1300
+ f"run {run_id}: could not confirm a live engine (unverifiable pid) — "
1301
+ "the request stands and fires if one is running",
1302
+ severity="warning",
1303
+ )
1304
+ return
1305
+ self.call_from_thread(
1306
+ self.notify,
1307
+ f"graceful stop requested — run {run_id} will stop after the current item "
1308
+ f"completes; continue later with `froid-loop resume {run_id}`",
1309
+ )
1310
+
1311
+ def action_delete_run(self) -> None:
1312
+ selected = self._selected_run_dir()
1313
+ if selected is None:
1314
+ return
1315
+ run_id, run_dir = selected
1316
+ # 'unknown' (a live-but-unreadable pid) does not block cleanup — see the
1317
+ # deliberate runs.engine_alive invariant — but the irreversible confirm
1318
+ # must not imply the run is safely dead, so it says so.
1319
+ live = data.liveness(run_dir)
1320
+ if live == "alive":
1321
+ self.notify(f"run {run_id} is live — stop it first", severity="warning")
1322
+ return
1323
+ warning = "this cannot be undone"
1324
+ if live == "unknown":
1325
+ warning = f"engine may still be live (unverifiable pid) — {warning}"
1326
+
1327
+ def done(ok: bool | None) -> None:
1328
+ if ok:
1329
+ self._delete_run_worker(run_id, run_dir)
1330
+
1331
+ self.push_screen(
1332
+ ConfirmModal(
1333
+ "delete run",
1334
+ f"permanently delete run {run_id}?",
1335
+ confirm_label="delete",
1336
+ warning=warning,
1337
+ ),
1338
+ done,
1339
+ )
1340
+
1341
+ @work(thread=True, group="lifecycle")
1342
+ def _delete_run_worker(self, run_id: str, run_dir: Path) -> None:
1343
+ try:
1344
+ runs.delete_run(self.project, run_dir)
1345
+ except (OSError, runs.LiveSessionError) as e:
1346
+ # LiveSessionError is the #419 backstop: the confirm above gates on engine
1347
+ # liveness, which an orphaned session passes. Surface it like any other
1348
+ # failed removal rather than letting it kill the worker thread.
1349
+ self.call_from_thread(self.notify, f"delete failed: {e}", severity="error")
1350
+ return
1351
+ self.call_from_thread(self._dashboard.forget_run, run_id)
1352
+ self.call_from_thread(self.notify, f"run {run_id} deleted")
1353
+
1354
+ def action_archive_run(self) -> None:
1355
+ selected = self._selected_run_dir()
1356
+ if selected is None:
1357
+ return
1358
+ run_id, run_dir = selected
1359
+ live = data.liveness(run_dir)
1360
+ if live == "alive":
1361
+ self.notify(f"run {run_id} is live — stop it first", severity="warning")
1362
+ return
1363
+
1364
+ def done(ok: bool | None) -> None:
1365
+ if ok:
1366
+ self._archive_run_worker(run_id, run_dir)
1367
+
1368
+ self.push_screen(
1369
+ ConfirmModal(
1370
+ "archive run",
1371
+ f"archive run {run_id} to .froid-loop/archive?",
1372
+ confirm_label="archive",
1373
+ warning=(
1374
+ "engine may still be live (unverifiable pid)" if live == "unknown" else None
1375
+ ),
1376
+ ),
1377
+ done,
1378
+ )
1379
+
1380
+ @work(thread=True, group="lifecycle")
1381
+ def _archive_run_worker(self, run_id: str, run_dir: Path) -> None:
1382
+ try:
1383
+ dest = runs.archive_run(self.project, run_dir)
1384
+ except (OSError, runs.LiveSessionError) as e:
1385
+ # see _delete_run_worker: the confirm's guard is engine-keyed, this one
1386
+ # is session-keyed (#419).
1387
+ self.call_from_thread(self.notify, f"archive failed: {e}", severity="error")
1388
+ return
1389
+ self.call_from_thread(self._dashboard.forget_run, run_id)
1390
+ self.call_from_thread(self.notify, f"run {run_id} archived to {dest}")
1391
+
1392
+ def action_cleanup_sessions(self) -> None:
1393
+ if self._mux_missing():
1394
+ return
1395
+
1396
+ def done(ok: bool | None) -> None:
1397
+ if ok:
1398
+ self._cleanup_sessions_worker()
1399
+
1400
+ self.push_screen(
1401
+ ConfirmModal(
1402
+ "cleanup sessions",
1403
+ "remove tmux sessions/windows for finished & stopped runs?",
1404
+ confirm_label="cleanup",
1405
+ ),
1406
+ done,
1407
+ )
1408
+
1409
+ @work(thread=True, group="lifecycle")
1410
+ def _cleanup_sessions_worker(self) -> None:
1411
+ # killed and unknown come from prune_sessions' single partition sample,
1412
+ # so the warning below only ever names sessions that were actually pruned.
1413
+ #
1414
+ # Guarded for the same reason as the ctl-window arm below, with the
1415
+ # opposite conclusion. This half is raiser-side too — the psmux backend
1416
+ # refuses a registry root that would fail its pre-spawn absoluteness gate,
1417
+ # and that raise is thrown before the tolerant listing wrapper can degrade
1418
+ # it — and an escape from a worker thread takes the whole dashboard down
1419
+ # (Textual's exit_on_error). Every CLI surface turns that same raise into
1420
+ # one named error through main()'s backstop; a worker thread has none.
1421
+ # But nothing has been killed yet, so there is no completed work to
1422
+ # protect: toast and stop, rather than carry on reporting a sweep that
1423
+ # never ran.
1424
+ try:
1425
+ killed, _live, unknown = runs.prune_sessions(self.project)
1426
+ except (MultiplexerError, UnicodeError) as e:
1427
+ self.call_from_thread(self.notify, f"session prune failed: {e}", severity="error")
1428
+ return
1429
+ # prune_ctl_windows probes has_session on the shared ctl session, a
1430
+ # raiser-side call; on a worker thread the toast must be marshalled, and
1431
+ # notify() must not be called directly (see _mux_guarded — foreground only).
1432
+ try:
1433
+ windows, survived, unverifiable = launch.prune_ctl_windows(self.project)
1434
+ except (MultiplexerError, UnicodeError) as e:
1435
+ # UnicodeError: a strict-POSIX decode fault from a scan probe that
1436
+ # does not normalize it to the seam type (#380) — the cli cleanup
1437
+ # arm's twin; an escape here kills the worker thread instead.
1438
+ # prune_sessions already killed the agent sessions above; surface the
1439
+ # ctl-window failure but keep reporting that completed work (and the
1440
+ # unknown-pid warning) rather than swallowing it on an early return.
1441
+ # Named: a bare transport message next to a "removed N session(s), 0
1442
+ # window(s)" toast reads as a successful window sweep.
1443
+ self.call_from_thread(self.notify, f"ctl window prune failed: {e}", severity="error")
1444
+ windows, survived, unverifiable = [], [], []
1445
+ if unknown:
1446
+ self.call_from_thread(
1447
+ self.notify,
1448
+ f"{len(unknown)} pruned session(s) had an unverifiable engine pid "
1449
+ f"(may still be live): {', '.join(sorted(unknown))}",
1450
+ severity="warning",
1451
+ )
1452
+ # The cli cleanup arm's stderr line, as a toast: the removal count below
1453
+ # excludes sessions the migration pass declined to claim in a legacy
1454
+ # registry, and a count that quietly excludes them reads as "all clean".
1455
+ # Read after the prune, so it describes what is left standing. Silent on
1456
+ # every platform without a registry namespace.
1457
+ # One toast per registry, naming it: there is more than one legacy
1458
+ # registry (psmux's default, and any root this process displaced), and
1459
+ # the operator's next action is to open the one holding these.
1460
+ for registry, names in runs.legacy_registry_leftovers(self.project).items():
1461
+ self.call_from_thread(
1462
+ self.notify,
1463
+ f"{len(names)} session(s) left in {registry} (not migrated): "
1464
+ f"{', '.join(names)} — see docs/multiplexer-backends.md before "
1465
+ "removing any of them",
1466
+ severity="warning",
1467
+ )
1468
+ # A kill that did not verifiably land gets its own toast rather than a
1469
+ # silent subtraction from the count below (#435) — the count now reports
1470
+ # only verified removals, so without this the windows would just vanish
1471
+ # from the report. Kept apart because they are different claims: one is
1472
+ # positive evidence the window is still there, the other is the absence
1473
+ # of any evidence at all. Both are retried by the next cleanup.
1474
+ if survived:
1475
+ self.call_from_thread(
1476
+ self.notify,
1477
+ f"{len(survived)} ctl window(s) still open after the kill: {', '.join(survived)}",
1478
+ severity="warning",
1479
+ )
1480
+ if unverifiable:
1481
+ self.call_from_thread(
1482
+ self.notify,
1483
+ f"{len(unverifiable)} ctl window(s) kill attempted, outcome unverifiable: "
1484
+ f"{', '.join(unverifiable)}",
1485
+ severity="warning",
1486
+ )
1487
+ self.call_from_thread(
1488
+ self.notify,
1489
+ f"removed {len(killed)} session(s), {len(windows)} window(s)",
1490
+ )
1491
+
1492
+ def action_validate(self) -> None:
1493
+ self._show_validate()
1494
+
1495
+ @work(thread=True, exclusive=True, group="captured")
1496
+ def _show_validate(self) -> None:
1497
+ """Preflight in a findings modal, degrading to the text one (#210).
1498
+
1499
+ A sibling of _show_captured rather than a change to it: that worker still
1500
+ serves the two dry runs, which have no document to parse.
1501
+
1502
+ The transport is the subprocess and `--json`, not documents.py's builders
1503
+ in-process, which its module docstring otherwise asks a non-CLI frontend
1504
+ to prefer. Knowing exception: cmd_validate imports third-party mux entry
1505
+ points and probes httpx, so the subprocess quarantines a broken plugin's
1506
+ import side effects and leaves the TUI's own lru_cached mux selection
1507
+ undisturbed. Extracting an in-process builder is a follow-up.
1508
+
1509
+ The body is guarded because @work(thread=True) defaults to
1510
+ exit_on_error=True: a JSONDecodeError or a KeyError escaping here would
1511
+ take the whole app down, not just this modal. exit_on_error=False is not
1512
+ the fix — that trades the crash for pressing `v` and nothing happening.
1513
+
1514
+ The degrade **re-runs validate in text mode** rather than showing the
1515
+ captured JSON. Dumping the document would withhold a perfectly good human
1516
+ rendering at the exact moment the structural one failed, and hand the
1517
+ reader a wall of `{"schema_version": ...}` instead. One sub-second
1518
+ subprocess on a path that should never fire buys a degrade that is
1519
+ byte-for-byte the pre-#210 behavior.
1520
+
1521
+ That re-run goes through _run_captured_guarded rather than calling
1522
+ run_captured directly: the except above does not cover it, and the two
1523
+ legs spawn the same subprocess, so a failure to spawn at all is not a
1524
+ JSON-leg failure a text re-run recovers from — it is the same failure
1525
+ twice, the second one escaping into exit_on_error.
1526
+ """
1527
+ tail = ["validate", "--project", str(self.project)]
1528
+ try:
1529
+ _rc, out, _err = launch.run_captured_streams([*tail, "--json"])
1530
+ doc = widgets.validate_document(out)
1531
+ except Exception: # a JSON-leg failure degrades, never kills the app
1532
+ doc = None
1533
+ if doc is None:
1534
+ rc, merged = self._run_captured_guarded(tail)
1535
+ screen = TextOutputModal("validate", rc, merged)
1536
+ else:
1537
+ screen = ValidateFindingsModal(doc)
1538
+ self.call_from_thread(self.push_screen, screen)
1539
+
1540
+ def _run_captured_guarded(self, tail: list[str]) -> tuple[int, str]:
1541
+ """run_captured, with a failure to spawn rendered as output, not raised.
1542
+
1543
+ Every caller is a @work(thread=True) body, and that decorator defaults to
1544
+ exit_on_error=True: an OSError out of subprocess.run — a deleted venv
1545
+ under sys.executable, EAGAIN off a loaded process table — would escape
1546
+ the worker and take the whole app down rather than this one modal.
1547
+
1548
+ The reason goes in the body rather than a notify() because the modal is
1549
+ already opening; a blank panel over an `exit 1` header would say only
1550
+ that something went wrong. The header carries which command it was, so
1551
+ the body does not repeat it.
1552
+ """
1553
+ try:
1554
+ return launch.run_captured(tail)
1555
+ except Exception as exc: # a failed spawn is a modal, not a crash
1556
+ return 1, f"could not run: {exc}"
1557
+
1558
+ @work(thread=True, exclusive=True, group="captured")
1559
+ def _show_captured(self, title: str, tail: list[str]) -> None:
1560
+ rc, out = self._run_captured_guarded(tail)
1561
+ self.call_from_thread(self.push_screen, TextOutputModal(title, rc, out))
1562
+
1563
+ def action_settings(self) -> None:
1564
+ if isinstance(self.screen, SettingsScreen):
1565
+ return
1566
+ try:
1567
+ doc = PolicyDoc.load(self.project / POLICY_FILE)
1568
+ except ParseError as e:
1569
+ self.notify(f"policy.toml is not valid TOML: {e}", severity="error")
1570
+ return
1571
+ self.push_screen(SettingsScreen(self.project, doc))
1572
+
1573
+
1574
+ def run_tui(project: Path) -> int:
1575
+ # Trip the once-per-process forced-backend warning while stderr is still the
1576
+ # real terminal: Textual captures sys.stderr for the app's whole run, so a
1577
+ # first firing inside the app (any observer gate) would consume the single
1578
+ # emission invisibly. Selection errors stay loud at their real call sites.
1579
+ try:
1580
+ mux_usable()
1581
+ except MultiplexerError:
1582
+ pass
1583
+ FroidLoopApp(project).run()
1584
+ return 0