froid-loop 0.11.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. froid_loop/__init__.py +11 -0
  2. froid_loop/__main__.py +12 -0
  3. froid_loop/adapters/__init__.py +3 -0
  4. froid_loop/adapters/base.py +254 -0
  5. froid_loop/adapters/entrypoints.py +63 -0
  6. froid_loop/adapters/env_fault.py +290 -0
  7. froid_loop/adapters/generic.py +2013 -0
  8. froid_loop/adapters/mock.py +49 -0
  9. froid_loop/adapters/multiplexer.py +914 -0
  10. froid_loop/adapters/opencode_http.py +1687 -0
  11. froid_loop/adapters/profile.py +650 -0
  12. froid_loop/adapters/psmux_backend.py +1428 -0
  13. froid_loop/adapters/registry.py +322 -0
  14. froid_loop/adapters/tmux_backend.py +35 -0
  15. froid_loop/adapters/tmux_base.py +630 -0
  16. froid_loop/checks.py +187 -0
  17. froid_loop/cli.py +5041 -0
  18. froid_loop/data/__init__.py +0 -0
  19. froid_loop/data/froid_loop_hook.py +228 -0
  20. froid_loop/data/froid_loop_probe_hook.py +88 -0
  21. froid_loop/data/plugins/example/plugin.toml +21 -0
  22. froid_loop/data/plugins/tea/plugin.toml +184 -0
  23. froid_loop/data/plugins/tea/tea_plugin.py +258 -0
  24. froid_loop/data/plugins/unity/plugin.toml +140 -0
  25. froid_loop/data/plugins/unity/unity_assets/FroidLoop.Unity.Editor.asmdef +16 -0
  26. froid_loop/data/plugins/unity/unity_assets/FroidLoop.Unity.Editor.asmdef.meta +7 -0
  27. froid_loop/data/plugins/unity/unity_assets/SceneAutoSaveGuard.cs +221 -0
  28. froid_loop/data/plugins/unity/unity_assets/SceneAutoSaveGuard.cs.meta +11 -0
  29. froid_loop/data/plugins/unity/unity_assets/_folders/Editor.meta +8 -0
  30. froid_loop/data/plugins/unity/unity_assets/_folders/FroidLoop.meta +8 -0
  31. froid_loop/data/plugins/unity/unity_cleanup.py +125 -0
  32. froid_loop/data/plugins/unity/unity_dialog_probe.py +239 -0
  33. froid_loop/data/plugins/unity/unity_facts.md +17 -0
  34. froid_loop/data/plugins/unity/unity_plugin.py +415 -0
  35. froid_loop/data/plugins/unity/unity_quiesce.py +234 -0
  36. froid_loop/data/plugins/unity/unity_ready.py +230 -0
  37. froid_loop/data/plugins/unity/unity_seed_assets.py +298 -0
  38. froid_loop/data/plugins/unity/unity_setup.py +551 -0
  39. froid_loop/data/plugins/unity/unity_teardown.py +362 -0
  40. froid_loop/data/profiles/antigravity.toml +52 -0
  41. froid_loop/data/profiles/claude.toml +85 -0
  42. froid_loop/data/profiles/codex.toml +22 -0
  43. froid_loop/data/profiles/copilot.toml +52 -0
  44. froid_loop/data/profiles/gemini.toml +26 -0
  45. froid_loop/data/profiles/opencode.toml +54 -0
  46. froid_loop/data/settings/core.toml +458 -0
  47. froid_loop/data/skills/README.md +93 -0
  48. froid_loop/data/skills/froid-loop-resolve/SKILL.md +288 -0
  49. froid_loop/data/skills/froid-loop-setup/SKILL.md +161 -0
  50. froid_loop/data/skills/froid-loop-setup/assets/module-help.csv +3 -0
  51. froid_loop/data/skills/froid-loop-setup/assets/module.yaml +19 -0
  52. froid_loop/data/skills/froid-loop-sweep/SKILL.md +100 -0
  53. froid_loop/data/skills/froid-loop-sweep/automation-mode.md +127 -0
  54. froid_loop/data/skills/froid-loop-sweep/deferred-work-format.md +302 -0
  55. froid_loop/data/skills/froid-loop-sweep/migration-mode.md +86 -0
  56. froid_loop/decisions.py +202 -0
  57. froid_loop/deferredwork.py +2282 -0
  58. froid_loop/devcontract.py +892 -0
  59. froid_loop/diagnostics.py +1104 -0
  60. froid_loop/documents.py +532 -0
  61. froid_loop/engine.py +7732 -0
  62. froid_loop/envvars.py +111 -0
  63. froid_loop/escalation.py +225 -0
  64. froid_loop/events.py +266 -0
  65. froid_loop/fences.py +103 -0
  66. froid_loop/froidconfig.py +226 -0
  67. froid_loop/frontmatter.py +526 -0
  68. froid_loop/gates.py +133 -0
  69. froid_loop/install.py +2936 -0
  70. froid_loop/journal.py +178 -0
  71. froid_loop/machine.py +148 -0
  72. froid_loop/model.py +898 -0
  73. froid_loop/operatoractions.py +474 -0
  74. froid_loop/platform_util.py +1490 -0
  75. froid_loop/plugins/__init__.py +64 -0
  76. froid_loop/plugins/bus.py +259 -0
  77. froid_loop/plugins/context.py +319 -0
  78. froid_loop/plugins/loader.py +145 -0
  79. froid_loop/plugins/manifest.py +279 -0
  80. froid_loop/plugins/model.py +296 -0
  81. froid_loop/plugins/registry.py +245 -0
  82. froid_loop/plugins/trust.py +75 -0
  83. froid_loop/policy.py +1569 -0
  84. froid_loop/probe.py +1044 -0
  85. froid_loop/process_host.py +408 -0
  86. froid_loop/recovery_flow.py +1561 -0
  87. froid_loop/resolve.py +283 -0
  88. froid_loop/runs.py +4715 -0
  89. froid_loop/runsetup.py +1293 -0
  90. froid_loop/sanitize.py +593 -0
  91. froid_loop/settings_schema.py +276 -0
  92. froid_loop/signals.py +160 -0
  93. froid_loop/sprintstatus.py +609 -0
  94. froid_loop/statemachine.py +57 -0
  95. froid_loop/stories.py +615 -0
  96. froid_loop/stories_engine.py +796 -0
  97. froid_loop/sweep.py +1892 -0
  98. froid_loop/tokens.py +196 -0
  99. froid_loop/tui/__init__.py +11 -0
  100. froid_loop/tui/app.py +1584 -0
  101. froid_loop/tui/data.py +840 -0
  102. froid_loop/tui/launch.py +1003 -0
  103. froid_loop/tui/screens/__init__.py +1 -0
  104. froid_loop/tui/screens/dashboard.py +1071 -0
  105. froid_loop/tui/screens/modals.py +943 -0
  106. froid_loop/tui/screens/settings_screen.py +477 -0
  107. froid_loop/tui/settings.py +135 -0
  108. froid_loop/tui/widgets.py +981 -0
  109. froid_loop/verify.py +4545 -0
  110. froid_loop/workspace.py +320 -0
  111. froid_loop/worktree_flow.py +2301 -0
  112. froid_loop-0.11.1.dist-info/METADATA +728 -0
  113. froid_loop-0.11.1.dist-info/RECORD +116 -0
  114. froid_loop-0.11.1.dist-info/WHEEL +4 -0
  115. froid_loop-0.11.1.dist-info/entry_points.txt +2 -0
  116. froid_loop-0.11.1.dist-info/licenses/LICENSE +30 -0
froid_loop/runs.py ADDED
@@ -0,0 +1,4715 @@
1
+ """Run-directory discovery and helpers shared by the CLI and the TUI."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import contextlib
6
+ import contextvars
7
+ import hashlib
8
+ import json
9
+ import math
10
+ import os
11
+ import re
12
+ import secrets
13
+ import shutil
14
+ import stat
15
+ import sys
16
+ import tarfile
17
+ import time
18
+ from collections.abc import Iterable, Mapping
19
+ from dataclasses import dataclass
20
+ from pathlib import Path
21
+ from typing import Any, Literal
22
+
23
+ from . import devcontract, envvars, verify
24
+ from .adapters.multiplexer import (
25
+ MultiplexerError,
26
+ TerminalMultiplexer,
27
+ get_multiplexer,
28
+ mux_usable,
29
+ )
30
+ from .frontmatter import auto_dev_baseline_of, parse_frontmatter, status_of
31
+ from .journal import STATE_FILE, VERIFY_DIR, Journal, load_state, save_state
32
+ from .model import PAUSE_ESCALATION, Phase, RunState, StoryTask
33
+ from .platform_util import (
34
+ MAX_SEGMENT,
35
+ UnconfinedWriteError,
36
+ _mkstemp_beside,
37
+ atomic_replace,
38
+ atomic_write_bytes_confined,
39
+ atomic_write_text_confined,
40
+ create_exclusive_confined,
41
+ has_parent_ref,
42
+ is_absolute_path,
43
+ is_link_like,
44
+ names_tree_root,
45
+ retrying_unlink,
46
+ safe_segment,
47
+ )
48
+ from .process_host import ProcessHostError, get_process_host
49
+
50
+ # The multiplexer registry's directory name inside a project's state subtree (see
51
+ # `mux_registry_root`). It sits beside the run entries and must never BE one: the
52
+ # leading underscore is what makes that structural, since `RUN_ID_RE` requires an
53
+ # alphanumeric first character, so no `--run-id` can key its state dir onto the
54
+ # registry. That is also what lets the orphan-state sweep tell the two apart by
55
+ # name alone (see `reconcile_orphan_state_dirs`).
56
+ MUX_REGISTRY_DIR = "_mux"
57
+ # psmux's own registry-root variable. Named here, in transport-agnostic code, for
58
+ # the same reason `PROJECT_OPTION` is: the export has to happen ahead of backend
59
+ # selection, which probes a subprocess, so it cannot be routed through a backend
60
+ # instance. See `export_psmux_registry_root`.
61
+ PSMUX_DATA_DIR = "PSMUX_DATA_DIR"
62
+ RUNS_DIR = Path(".froid-loop") / "runs"
63
+ ARCHIVE_DIR = Path(".froid-loop") / "archive"
64
+ PID_FILE = "engine.pid"
65
+ # Cross-process channel for a stop request: a control file the requester (CLI/TUI)
66
+ # writes and the engine reads. The body carries a `mode` — "graceful" or "hard".
67
+ #
68
+ # graceful (`stop --graceful`): finish the in-flight item, then finalize and stop.
69
+ # Honored at item boundaries only; resumable.
70
+ # hard (`stop`): stop now. Lodged by stop_run *before* it signals, honored by the
71
+ # engine at item boundaries and mid-session by the adapter wait loop.
72
+ #
73
+ # The file exists because signals are not a portable stop channel: there is no
74
+ # SIGUSR1 on Windows/psmux, and an inter-process SIGTERM is never delivered to a
75
+ # native-Windows engine at all, so the win32 "graceful" terminate is a no-op that
76
+ # only ever burned _STOP_WAIT_S into a force-kill (#319). SIGTERM remains the POSIX
77
+ # fast path — the file is what makes a stop work everywhere else. The engine stays
78
+ # the single writer of journal.jsonl, and the single *consumer* of this file;
79
+ # requesters only ever write it, adapters only ever read it.
80
+ STOP_REQUEST_FILE = "stop-request.json"
81
+ # The host-exec config baseline's name inside a run's state dir (see
82
+ # `config_digest_path_for`). A bare hex digest, not JSON: one opaque token, and a
83
+ # format an operator can read with `cat`.
84
+ CONFIG_DIGEST_FILE = "config-digest"
85
+ # Read cap for the file above. A sha256 hex digest is 64 bytes; the slack is for
86
+ # a trailing newline and for saying "this is not the digest" out of a file that
87
+ # is merely wrong rather than hostile. The cap's real job is the hostile case —
88
+ # see `read_trusted_config_digest` on why a bound, not a bigger buffer.
89
+ _MAX_DIGEST_BYTES = 256
90
+ _INVALID_PID_IDENTITY = -1.0 # impossible process start/create time; forces "not ours"
91
+
92
+
93
+ class StopRunError(Exception):
94
+ """A live run could not be stopped — the engine honored neither channel (the
95
+ lodged stop request nor SIGTERM) and its pid's identity can no longer be
96
+ verified, so force-killing would risk an unrelated (reused) pid. The caller
97
+ surfaces this rather than silently marking stopped."""
98
+
99
+
100
+ class GracefulStopError(Exception):
101
+ """A graceful-stop request could not be lodged (run already finished, or its
102
+ engine is provably dead so the request would never be consumed). ``str()`` is
103
+ the operator-facing message the CLI/TUI surface verbatim."""
104
+
105
+
106
+ class LiveSessionError(Exception):
107
+ """A run directory was not removed because the run's agent session is still
108
+ live (see :func:`live_session_may_be_ours`). ``str()`` is the operator-facing
109
+ message the CLI/TUI surface verbatim."""
110
+
111
+
112
+ # How long stop_run waits for a signalled engine to exit before falling back to
113
+ # marking the run stopped itself.
114
+ _STOP_WAIT_S = 10.0
115
+ _STOP_POLL_S = 0.1
116
+ # How long stop_run lets a force-kill settle before deciding it failed. A kill that
117
+ # returns cleanly is not proof of death — win32 shells `taskkill /F /T` with
118
+ # `check=False`, so a refused kill raises nothing — but the pid can also linger for a
119
+ # moment after a delivered SIGKILL, and `is_alive` is a bare existence probe that
120
+ # reads a not-yet-reaped process as alive. Long enough to outlast that, short enough
121
+ # that a genuinely surviving engine is still noticed while the operator waits.
122
+ _KILL_CONFIRM_S = 0.5
123
+
124
+
125
+ def new_run_id() -> str:
126
+ return time.strftime("%Y%m%d-%H%M%S") + "-" + secrets.token_hex(2)
127
+
128
+
129
+ # A run id is a lookup key with exactly one legitimate producer (new_run_id), and it
130
+ # lands in three positions at once: a directory name under RUNS_DIR, a multiplexer
131
+ # session name (froid-loop-<id>), and a git ref component (froid-loop/<id>/<unit>).
132
+ # So an id supplied from outside is *rejected*, never sanitized — coercing it would
133
+ # break the id<->path<->session bijection the CLI relies on to find a run again.
134
+ #
135
+ # The charset is a superset of every new_run_id() output and excludes, by
136
+ # construction: path separators and `..` (traversal), `<>:"|?*` plus trailing dots
137
+ # and spaces (Windows), `.` and `:` (multiplexer session-name mangling), and all
138
+ # whitespace/control characters. It is also identity under safe_ref_segment, so the
139
+ # unit branch a run produces reads back verbatim — hence no ref check below.
140
+ RUN_ID_RE = re.compile(r"[A-Za-z0-9][A-Za-z0-9_-]*")
141
+
142
+
143
+ def is_valid_run_id(value: str) -> bool:
144
+ """True when ``value`` is a run id we would have produced ourselves — the guard
145
+ every externally-supplied ``--run-id`` and every id recomposed from the outside
146
+ world (a foreign multiplexer session name) must pass before it touches a path.
147
+
148
+ The length cap is ``platform_util.MAX_SEGMENT``: a run id is a directory name.
149
+ The ``safe_segment`` identity check adds the one rule ``RUN_ID_RE`` cannot
150
+ express — the reserved Windows device basenames (``CON``, ``NUL``, ``COM1``…),
151
+ which are legal-looking ids that no filesystem will accept as a directory.
152
+
153
+ The control-session shape (``ctl``, ``ctl-…``, any letter case) is reserved
154
+ on the same principle, against the multiplexer namespace instead of the
155
+ filesystem's — see :func:`is_reserved_run_id` for the shape and why case is
156
+ folded. Refusing the id here is what makes the two session namespaces
157
+ disjoint: every agent session is ``froid-loop-<valid id>``, so none can
158
+ reach the control session's name."""
159
+ return _wellformed_run_id(value) and not is_reserved_run_id(value)
160
+
161
+
162
+ def _wellformed_run_id(value: str) -> bool:
163
+ """The shape half of :func:`is_valid_run_id`: charset, length, and the
164
+ reserved-device-basename identity check — everything except the
165
+ control-session reservation. Split out because the *parse* side
166
+ (:func:`_agent_run_id`) must accept ids the *mint* refuses: a run
167
+ persisted by an older release under e.g. ``ctl-foo`` owns a genuine
168
+ ``froid-loop-ctl-foo`` agent session that the sweep has to be able to
169
+ reach."""
170
+ return (
171
+ bool(RUN_ID_RE.fullmatch(value))
172
+ and len(value) <= MAX_SEGMENT
173
+ and safe_segment(value) == value
174
+ )
175
+
176
+
177
+ def is_reserved_run_id(value: str) -> bool:
178
+ """The MINT-side reservation: any id of the control-session shape (``ctl``
179
+ or ``ctl-…``, any letter case) is refused at :func:`is_valid_run_id`.
180
+ Deliberately broader than :func:`run_id_aliases_control_session` — a new id
181
+ anywhere near the control namespace buys nothing but confusion, so none is
182
+ admitted — while the read paths, which must handle ids an older release
183
+ already persisted, use the narrow test. ``RUN_ID_RE`` is ASCII-only, so
184
+ ``str.lower`` is the exact fold (see the narrow test for why case folds at
185
+ all)."""
186
+ v = value.lower()
187
+ return v == "ctl" or v.startswith("ctl-")
188
+
189
+
190
+ def run_id_aliases_control_session(value: str) -> bool:
191
+ """True when ``session_name(value)`` names a session that can BE a live
192
+ control session: the fixed name (id ``ctl``) or a per-registry digest name
193
+ (id ``ctl-<16 hex>`` — the only suffix :func:`ctl_session_for` can mint).
194
+ The adapter's ensure-session would *adopt* that live session as the run's
195
+ own, and the run's teardown would kill the whole control session, every
196
+ parked window of every run in it — so the project-free READ paths key on
197
+ this: :func:`kill_session` skips such an id, :func:`_agent_run_id`
198
+ refuses to read such a session as a run, and ``cli``/the TUI refuse to
199
+ resume/re-arm/replan such a run. This is the SHAPE question — "could
200
+ this name be a control session's on some registry" — and it must stay
201
+ out of any site asking the *instance* question ("is it the control
202
+ session this process addresses"): :func:`live_session_may_be_ours`
203
+ compares against the actual names (the fixed one plus this project's
204
+ :func:`ctl_session_for`), because discounting the whole shape there
205
+ destroyed run dirs under live `ctl-<other digest>` agents on tmux.
206
+
207
+ Compared **case-insensitively**: psmux resolves a session by opening
208
+ ``<data dir>\\<name>.port`` by name (``src/paths.rs:113``, source-read at
209
+ v3.3.8), and NTFS opens names case-insensitively — measured: with
210
+ ``froid-loop-ctl-x`` live, target ``froid-loop-CTL-x`` answers
211
+ ``has-session``, is refused as a duplicate by ``new-session``, and a kill
212
+ through it takes the lowercase session down.
213
+
214
+ Deliberately narrower than :func:`is_reserved_run_id`: a historical
215
+ ``ctl-foo`` run's session is a GENUINE agent session, distinct from every
216
+ control session and addressable exactly and safely (tmux: measured, the
217
+ exact full target removes only it; our seam sends ``=``-exact targets —
218
+ ``tmux_base.py:141,166``, source-read. psmux: exact port files, case
219
+ aside). Skipping those too made such runs unreachable by ``stop`` and
220
+ ``cleanup`` both. Ceiling, named: an id of exactly the digest shape whose
221
+ hex is NOT the current registry's digest is also skipped — undecidable
222
+ without the project in hand, and the leak direction (one stale session
223
+ left standing) is the safe one."""
224
+ return is_ctl_session_name(session_name(value).lower())
225
+
226
+
227
+ def is_parsable_run_id(value: str) -> bool:
228
+ """The PARSE-side counterpart of :func:`is_valid_run_id`: may an id
229
+ recovered from an existing multiplexer name be acted on as a run?
230
+
231
+ The two questions are different and must never share a predicate.
232
+ :func:`is_valid_run_id` answers "may a NEW id be this", so it carries the
233
+ mint's broad ctl reservation (:func:`is_reserved_run_id`) — and a reader
234
+ that borrows it stops recognising every id an older release already
235
+ persisted. A ``ctl-foo`` run minted before that reservation owns a real
236
+ run dir and a real ``run-ctl-foo`` control-session window; asking the
237
+ mint's question about them leaks both, unreachable by the sweep forever.
238
+
239
+ So: the shape half (:func:`_wellformed_run_id` — charset, length, and the
240
+ reserved-device-basename check, because the id still steers a run-dir
241
+ path) minus only the narrow alias test
242
+ (:func:`run_id_aliases_control_session`), which the read paths key on
243
+ because reading one of THOSE as a run points a kill path at the control
244
+ plane. Exactly :func:`_agent_run_id`'s guard, public so the other parse
245
+ sites ask it instead of re-deriving it — the ctl-window sweep in
246
+ ``tui.launch`` did borrow the mint's, and parked pre-upgrade windows
247
+ leaked from ``cleanup`` because of it."""
248
+ return _wellformed_run_id(value) and not run_id_aliases_control_session(value)
249
+
250
+
251
+ def list_run_dirs(project: Path) -> list[Path]:
252
+ """All run dirs containing a state.json, oldest first (run ids sort
253
+ chronologically)."""
254
+ runs = project / RUNS_DIR
255
+ if not runs.is_dir():
256
+ return []
257
+ return sorted(d for d in runs.iterdir() if (d / "state.json").is_file())
258
+
259
+
260
+ def all_run_dirs(project: Path) -> list[Path] | None:
261
+ """Every run dir under the runs root — ``state.json`` or not — oldest first,
262
+ or ``None`` when the listing could not be taken.
263
+
264
+ The ungated counterpart to :func:`list_run_dirs`, and the one to ask when the
265
+ question is "does a run still own its control plane" rather than "which runs
266
+ can I read". A run whose state.json was removed or corrupted still holds a
267
+ live ``engine.pid``, so the gated view walks straight past exactly the run an
268
+ operator is mid-recovery on — the hazard :func:`_run_dir_names` documents,
269
+ whose set this wraps rather than re-listing.
270
+
271
+ ``None`` is an unreadable runs root and means *nothing was learned*, which is
272
+ not the same answer as the empty list a missing root gives. Callers that act
273
+ on "no live runs" have to tell those apart; see :func:`_run_dir_names`.
274
+ """
275
+ names = _run_dir_names(project)
276
+ if names is None:
277
+ return None
278
+ root = project / RUNS_DIR
279
+ return sorted(root / name for name in names)
280
+
281
+
282
+ def latest_run_dir(project: Path) -> Path | None:
283
+ candidates = list_run_dirs(project)
284
+ return candidates[-1] if candidates else None
285
+
286
+
287
+ def write_named_pid(pidfile: Path, pid: int) -> None:
288
+ """Record ``pid`` plus its identity to ``pidfile``, so a later liveness read can
289
+ tell our process from a stranger that inherited a reused pid (immediate on
290
+ Windows). One whitespace-delimited line: ``"<pid>"`` (legacy) or
291
+ ``"<pid> <identity>"``; the identity token is omitted when the platform can't
292
+ provide one. The parameterized form :func:`write_pid` builds on — reused for the
293
+ Unity dialog probe's own ``unity-dialog-probe.pid`` handle."""
294
+ identity = get_process_host().identity(pid)
295
+ line = f"{pid} {identity}" if identity is not None else str(pid)
296
+ pidfile.write_text(line, encoding="utf-8")
297
+
298
+
299
+ def write_pid(run_dir: Path) -> None:
300
+ """Record the engine pid plus its identity, so a later liveness read can tell
301
+ our engine from a stranger that inherited a reused pid (immediate on Windows).
302
+ Never deleted: a stale pid that reads as gone is the signal a run was
303
+ interrupted."""
304
+ write_named_pid(run_dir / PID_FILE, os.getpid())
305
+
306
+
307
+ def session_name(run_id: str) -> str:
308
+ return f"froid-loop-{run_id}"
309
+
310
+
311
+ def attach_target_argv(target: str) -> list[str]:
312
+ """Multiplexer command to reach a target session/window (see
313
+ :meth:`TerminalMultiplexer.attach_target_argv`)."""
314
+ return get_multiplexer().attach_target_argv(target)
315
+
316
+
317
+ def session_target(run_id: str) -> str:
318
+ """Seam-canonical target token for the run's agent session (see
319
+ :meth:`TerminalMultiplexer.target`)."""
320
+ return get_multiplexer().target(session_name(run_id))
321
+
322
+
323
+ def attach_argv(run_id: str) -> list[str]:
324
+ return attach_target_argv(session_target(run_id))
325
+
326
+
327
+ # ------------------------------------------------------- user-scoped state root
328
+
329
+
330
+ class StateRootError(Exception):
331
+ """No user-scoped state root could be derived from this environment — every
332
+ candidate base was unset, empty, relative, or named the filesystem root. The
333
+ control plane has nowhere to live, and the caller must fail rather than guess
334
+ (see :func:`state_root`)."""
335
+
336
+
337
+ def _state_base(value: str | None) -> Path | None:
338
+ """``value`` as a usable base directory, or ``None`` when it cannot be one.
339
+
340
+ The single rule every *derived* candidate below is held to, so the POSIX and
341
+ win32 branches cannot drift into judging their inputs differently. A base is
342
+ rejected when it is unset, empty, relative, or names the filesystem root
343
+ itself. The last three are the answers a broken environment gives *instead* of
344
+ raising, which is what makes them worth naming:
345
+
346
+ - **empty**: ``os.path.expanduser("~")`` answers ``""`` on Windows for a
347
+ set-but-empty ``USERPROFILE``, and ``Path("")`` is the current directory.
348
+ - **relative**: including ``"~"`` itself, which is what ``expanduser`` returns
349
+ when it cannot expand at all. The state root would then move with the
350
+ launch cwd, and a run whose control plane it cannot find again is a run
351
+ that stalls to ``session_timeout_min`` rather than one that fails.
352
+ - **the root**: ``expanduser("~")`` answers ``"/"`` on POSIX for a set-but-empty
353
+ ``HOME`` (``posixpath`` folds the empty prefix to the root), which would put
354
+ ``/.local/state/froid-loop`` on the filesystem root — a permission error for
355
+ an ordinary user and, for a containerised root, a silent write to ``/``.
356
+ ``base == base.parent`` is the root test on both flavours.
357
+
358
+ ``os.path.isabs`` rather than :func:`platform_util.is_absolute_path`: the
359
+ latter is purpose-built for "must stay inside the project" guards and is
360
+ strictly broader — it calls the drive-*relative* ``C:foo`` absolute, which is
361
+ exactly the value that must not become a state root. The question here is the
362
+ platform's own, and each branch below only ever runs on its own platform.
363
+ """
364
+ if not value or not os.path.isabs(value):
365
+ return None
366
+ base = Path(value)
367
+ return None if base == base.parent else base
368
+
369
+
370
+ def state_root() -> Path:
371
+ """The froid-loop state root for this user: the out-of-tree home of per-run
372
+ control-plane state — the events channel (#494) and, later, the config digest
373
+ (#498). Outside the project tree because a branch switch, a worktree mount or
374
+ a rollback must not be able to take a live run's control plane away.
375
+
376
+ Resolution, first answer wins:
377
+
378
+ 1. ``FROID_LOOP_STATE_DIR``, used as the state root **itself** — no
379
+ ``froid-loop`` segment is appended, because the variable names our root
380
+ rather than a base to build one under. It is honoured as spelled (see
381
+ :func:`envvars.state_dir`) and is not passed through ``_state_base``:
382
+ *skipping* a stated override would be a silent countermand, where skipping
383
+ a derived base only moves on to the next guess.
384
+
385
+ It must still be **absolute**, and a relative spelling raises rather than
386
+ being resolved for the operator. Absoluteness is not a matter of taste
387
+ here — the root is read by two processes with different working
388
+ directories. The engine exports it to the session as
389
+ ``FROID_LOOP_EVENTS_DIR`` and the multiplexer launches that session at
390
+ ``spec.cwd`` (a worktree under isolation), while the watcher polls it from
391
+ the orchestrator's own cwd. A relative root therefore names two different
392
+ directories at once: the relay writes its Stop where nothing is watching,
393
+ and the run waits out ``session_timeout_min`` — the exact silent stall
394
+ ``_state_base`` rejects relative *derived* bases to avoid, and the one
395
+ this whole channel was moved out of the tree to prevent.
396
+
397
+ Raising is not the countermand the paragraph above refuses: it names the
398
+ variable and the fix, where absolutizing against whichever cwd this
399
+ process happens to have would be the guess. The not-the-root half of
400
+ ``_state_base``'s rule is deliberately *not* applied — that half exists to
401
+ stop a broken environment's ``""`` from landing a guess at ``/``, and an
402
+ override is not a guess.
403
+ 2. POSIX — ``$XDG_STATE_HOME/froid-loop`` when that variable names an absolute
404
+ path, else ``~/.local/state/froid-loop``. A relative ``XDG_STATE_HOME`` is
405
+ *ignored*, which the XDG base-directory spec requires of its consumers.
406
+ (``install._shield_inherited_excludes`` resolves a relative
407
+ ``XDG_CONFIG_HOME`` instead of ignoring it — the opposite call for the
408
+ opposite reason: there we reproduce *git's* reading of the variable, here
409
+ we are the spec's own consumer.)
410
+ 3. win32 — ``%LOCALAPPDATA%\\froid-loop\\state``, else
411
+ ``%USERPROFILE%\\AppData\\Local\\froid-loop\\state``. ``LOCALAPPDATA`` names
412
+ the per-user, per-machine, non-roaming store Windows intends for exactly
413
+ this, and the second form is its documented default location.
414
+
415
+ **Never** ``Path.home()`` on the win32 arm. It is ``ntpath.expanduser("~")``,
416
+ which prefers ``USERPROFILE`` and then falls back to ``HOMEDRIVE`` +
417
+ ``HOMEPATH`` — a pair that on a domain-joined machine may name a network home
418
+ share. A control plane whose atomic renames and ``O_NOFOLLOW``-anchored writes
419
+ live on an SMB share is not the local directory this needs, and the derivation
420
+ also disagrees with the one git uses for its own ``$HOME``
421
+ (``install._shield_home_git_ignore`` documents that split in full). Reading
422
+ ``LOCALAPPDATA``/``USERPROFILE`` directly asks for the store by name instead of
423
+ inferring it from a home.
424
+
425
+ Raises :class:`StateRootError` when no candidate answers. This is a write
426
+ path, so it raises rather than degrading to a plausible-looking default:
427
+ ``platform_util.resolve_or_lexical`` states the doctrine (observation may
428
+ degrade, repair writes must raise), and the degraded outcomes here are all
429
+ silent — a control plane at the cwd, or at ``/``, that the *next* process to
430
+ ask resolves somewhere else.
431
+ """
432
+ override = envvars.state_dir()
433
+ if override:
434
+ # `os.path.isabs` on the raw string, matching `_state_base` exactly rather
435
+ # than `Path.is_absolute` — the rule and its reason are stated there.
436
+ if not os.path.isabs(override):
437
+ raise StateRootError(
438
+ f"{envvars.STATE_DIR} must name an absolute directory: {override!r} is "
439
+ "relative, and the state root is read by both this process and the "
440
+ "session it launches — which run from different working directories, "
441
+ "so a relative root names two different places and the run's "
442
+ "completion signal is written where nothing is watching"
443
+ )
444
+ return Path(override)
445
+ if sys.platform == "win32":
446
+ local = _state_base(os.environ.get("LOCALAPPDATA"))
447
+ if local:
448
+ return local / "froid-loop" / "state"
449
+ profile = _state_base(os.environ.get("USERPROFILE"))
450
+ if profile:
451
+ return profile / "AppData" / "Local" / "froid-loop" / "state"
452
+ else:
453
+ xdg = _state_base(os.environ.get("XDG_STATE_HOME"))
454
+ if xdg:
455
+ return xdg / "froid-loop"
456
+ home = _state_base(os.path.expanduser("~"))
457
+ if home:
458
+ return home / ".local" / "state" / "froid-loop"
459
+ raise StateRootError(
460
+ "cannot locate a state directory for froid-loop's run control plane: "
461
+ + (
462
+ "neither %LOCALAPPDATA% nor %USERPROFILE% names an absolute directory"
463
+ if sys.platform == "win32"
464
+ else "neither $XDG_STATE_HOME nor $HOME names an absolute directory"
465
+ )
466
+ + f" — set {envvars.STATE_DIR} to the directory it should live in"
467
+ )
468
+
469
+
470
+ def project_state_root(project: Path) -> Path:
471
+ """The subtree of :func:`state_root` holding every run of this project:
472
+ ``<state root>/<project key>``. Split out from :func:`state_dir_for` because
473
+ the GC reads it as a *directory to enumerate* rather than composing one run's
474
+ path — see :func:`reconcile_orphan_state_dirs`, whose whole job is the entries
475
+ under here that no longer have a run dir."""
476
+ return state_root() / project_tag(project)
477
+
478
+
479
+ def mux_registry_root(project: Path) -> Path:
480
+ """This project's terminal-multiplexer registry root:
481
+ ``<state root>/<project key>/_mux`` (see :data:`MUX_REGISTRY_DIR`).
482
+
483
+ A *registry* is the directory a multiplexer keeps its per-session addressing
484
+ state in — psmux writes one ``.port``/``.key``/``.sid``/``.pid`` quartet per
485
+ session under ``PSMUX_DATA_DIR`` (default ``%USERPROFILE%\\.psmux``), and
486
+ every verb resolves a session by reading that quartet back. Two processes
487
+ that disagree about the root therefore disagree about which sessions exist,
488
+ which is why the root is *derived* — from the project, through the same
489
+ :func:`project_tag` every ownership tag already uses — rather than minted per
490
+ run, read from a file, or taken from whatever the launching shell exported.
491
+ See :func:`export_psmux_registry_root` for the export and its rules.
492
+
493
+ Keyed on the project rather than on froid-loop as a whole so a prune bug in
494
+ one project cannot address another project's servers at all: the partition
495
+ becomes structural instead of a filter (the ``@froid_project`` tag stays, as
496
+ the tmux-side answer and the belt). The price is that one ``psmux ls`` no
497
+ longer shows every froid-loop session on the machine — stated for the operator
498
+ in ``docs/multiplexer-backends.md`` and printed by ``froid-loop mux``.
499
+
500
+ Under :func:`state_root` and not in the project tree, deliberately: a branch
501
+ switch or a rollback that deleted a ``.port`` file would leave the server
502
+ alive, unreachable, and invisible to ``psmux ls`` in *any* registry — a
503
+ manufactured orphan. Same doctrine :func:`state_root` itself exists for.
504
+ """
505
+ return project_state_root(project) / MUX_REGISTRY_DIR
506
+
507
+
508
+ def export_psmux_registry_root(project: Path) -> str | None:
509
+ """Point this process — and everything it spawns — at ``project``'s registry
510
+ by exporting ``PSMUX_DATA_DIR``. Returns the value in force afterwards, or
511
+ ``None`` when no root could be derived.
512
+
513
+ **The process environment, not a per-call argument.** The seam spawns every
514
+ psmux verb through ``BaseTmuxBackend._run``, whose ``env=None`` default means
515
+ *inherit this process's environment*, and a create-call-only injection is
516
+ worse than none: the session's server would come up under a root every later
517
+ ``has_session`` / ``list_window_ids`` cannot see, and those verbs report an
518
+ unreadable registry as ``False`` / ``[]`` — a live run reading itself as gone.
519
+ One export ahead of dispatch covers every verb in-process.
520
+
521
+ **The root is always derived, and an ambient value never changes it.** That
522
+ is the whole rule, and the absence of an exception is the point:
523
+ :func:`mux_registry_root` is a pure function of (project, state root), so any
524
+ two froid-loop processes given the same project and the same state root agree
525
+ — which is the entire property #537 exists to establish. A value already in
526
+ the environment is *overridden*, and the caller says so
527
+ (:func:`cli._configure_mux` reports it once on stderr; ``froid-loop mux``
528
+ discloses it).
529
+
530
+ **Why an operator's own ``PSMUX_DATA_DIR`` is not honoured**, since honouring
531
+ it is the obvious kindness and it was tried:
532
+
533
+ - It would make the registry a function of the launch *shell*. A TUI started
534
+ from the Start menu carries no profile environment and derives; a run
535
+ started from a dev shell whose profile exports a root honours that root.
536
+ Two registries on one machine, and a live session reading as gone in one of
537
+ them — which is the failure this module exists to prevent, not a corner of
538
+ it.
539
+ - Whether honouring is even the right answer is unknowable from here. A
540
+ process that finds a root in its environment cannot tell one the operator
541
+ typed once in *this* shell — where a clean sibling process would derive —
542
+ from one their profile exports into *every* shell, where a clean sibling
543
+ honours it. The two produce byte-identical environments and want opposite
544
+ answers, so no comparison settles it: the missing fact is the operator's
545
+ intent, and it is not in the environment.
546
+ - It contradicts the promise made beside it. ``FROID_LOOP_STATE_DIR``'s
547
+ documentation says there is deliberately no second variable naming the
548
+ registry, because "two knobs that can disagree would put two processes on
549
+ different registries, each blind to the other's live sessions". An ambient
550
+ ``PSMUX_DATA_DIR`` is exactly that second knob.
551
+
552
+ Overridden rather than *refused*, deliberately: ``PSMUX_DATA_DIR`` is psmux's
553
+ variable, and an operator may have it set for their own sessions with no
554
+ thought of froid-loop at all. Erroring out of every command on such a machine
555
+ would be froid-loop claiming a name it does not own. The remedy runs the other
556
+ way and ``froid-loop mux`` prints it ready to paste: point *your* shell at
557
+ froid-loop's root, which is a function of the project rather than of whichever
558
+ shell happened to launch something.
559
+
560
+ **Overridden, but not abandoned.** A machine that had an absolute value
561
+ exported before the upgrade kept its froid-loop sessions in THAT registry,
562
+ because the old backend simply inherited it — so the displaced root is
563
+ handed to :func:`~.adapters.psmux_backend.note_displaced_registry` here,
564
+ the last moment anything can still read it, and the migration sweep runs a
565
+ tag-scoped pass over it alongside psmux's default
566
+ (:meth:`~.adapters.psmux_backend.PsmuxMultiplexer.legacy_registries`).
567
+ Without that the override would strand exactly the sessions it displaced,
568
+ with cleanup reporting a clean machine.
569
+
570
+ Wanting one registry to serve both is a real request and is deliberately not
571
+ answered here. It needs a stated operator preference rather than a guess at
572
+ one — and it must be a policy *whether*, never a *where*: ``policy.toml`` is
573
+ written by the sessions this orchestrator drives, so a policy-sourced root
574
+ would let a driven session choose which registry the cleanup path kills in.
575
+
576
+ **No ``FROID_LOOP_*`` knob for the root either.** It is derived state, not
577
+ configuration; ``FROID_LOOP_STATE_DIR`` already relocates it transitively —
578
+ one knob, one cascade, instead of two that can disagree. And ``envvars.py``
579
+ gains no entry for ``PSMUX_DATA_DIR`` itself: that module is scoped to
580
+ ``FROID_LOOP_*`` names and this is psmux's own, unregistered on the same
581
+ precedent as ``PSMUX_ALLOW_NESTING``.
582
+
583
+ **No root travels between processes.** Because every froid-loop process
584
+ derives its own root, nothing about a registry has to be transported at
585
+ all. What does have to travel is the *state root*: coding-CLI windows are
586
+ told it explicitly through their env dict (:func:`pinned_state_env`), and
587
+ everything else — a session's window-0 shell, the TUI's parked engine
588
+ windows — inherits it, as it always has. psmux's ``PSMUX_BARE_ENV=1`` mode
589
+ breaks that inheritance and is **not supported**: the psmux backend warns
590
+ once per process when it is on (see ``PsmuxMultiplexer._warn_if_bare_env``).
591
+
592
+ Never raises. This runs ahead of *every* command, ``diagnose`` and
593
+ ``validate`` included, and an underivable state root must not take the
594
+ diagnostics down with it. ``None`` means "no root established": psmux keeps
595
+ whatever it had, which is also the root cleanup sweeps as the legacy one.
596
+ """
597
+ try:
598
+ root = str(mux_registry_root(project))
599
+ except (StateRootError, OSError, RuntimeError):
600
+ # OSError/RuntimeError: project_tag resolves the project, which raises on
601
+ # a path the OS cannot canonicalize and, below 3.13, on a symlink loop.
602
+ # The ambient value is left exactly as found — there is nothing better to
603
+ # put there, and PsmuxMultiplexer._run still refuses to spawn under a
604
+ # value psmux would panic on.
605
+ return None
606
+ displaced = os.environ.get(PSMUX_DATA_DIR)
607
+ os.environ[PSMUX_DATA_DIR] = root
608
+ if displaced is not None and displaced != root:
609
+ # The variable is now gone, and it was the only record of where a
610
+ # pre-upgrade machine's sessions live: before #537 the backend simply
611
+ # inherited it. Hand it to the backend that has to sweep there, at the
612
+ # one moment it is still knowable. Imported here rather than at module
613
+ # scope because this is the psmux leaf, and this module talks to the
614
+ # seam — the coupling is confined to the function already named for
615
+ # psmux's own variable.
616
+ from .adapters.psmux_backend import note_displaced_registry
617
+
618
+ note_displaced_registry(displaced)
619
+ return root
620
+
621
+
622
+ def pinned_state_env() -> dict[str, str]:
623
+ """``{FROID_LOOP_STATE_DIR: <this process's state root>}``, for a child that
624
+ must land on the same one — or ``{}`` when no root can be derived.
625
+
626
+ A convenience spelling of :func:`pin_state_root` over an empty dict, for
627
+ composing env dicts (the engine's session env spreads it in). The final
628
+ merge before a window launch goes through :func:`pin_state_root` itself —
629
+ a spread of this dict is only an ordering guarantee, and ordering
630
+ guarantees nothing when the dict is ``{}``.
631
+
632
+ **Resolved, never forwarded.** Passing this only when the operator set it
633
+ would leave exactly the default case broken, which is the common one. What
634
+ travels is the answer this process reached, however it reached it.
635
+
636
+ What follows the state root, and what does not, since the two are easy to
637
+ swap: the run's control plane (:func:`state_dir_for`), its event channel
638
+ (:func:`events_dir_for`) and the multiplexer registry
639
+ (:func:`mux_registry_root`) all live under it, so a child computing a
640
+ different root writes and reads where nothing else looks. The run *directory*
641
+ does not — :func:`run_dir_for` is in-tree at ``<project>/.froid-loop/runs``
642
+ and moves with the project, not with this.
643
+
644
+ ``{}`` rather than a raise: a child told nothing derives its own answer and
645
+ fails on the same broken environment with its own message, which is better
646
+ than a launcher that cannot report anything at all.
647
+ """
648
+ return pin_state_root({})
649
+
650
+
651
+ def pin_state_root(env: Mapping[str, str]) -> dict[str, str]:
652
+ """``env`` with its ``FROID_LOOP_STATE_DIR`` entry forced to this process's
653
+ own answer: **set** to the resolved state root when one derives, **removed**
654
+ when none does. Other keys pass through untouched.
655
+
656
+ The chokepoint for every merge where a caller-supplied env (a profile's
657
+ ``[env]`` table rides those dicts) meets the state-root pin — the engine's
658
+ coding-CLI window, the probe window, and the attached resolve session. A
659
+ "pin spreads last" ordering rule is not enough, because with an underivable
660
+ state root there is no pin key to order: :func:`pinned_state_env` is ``{}``
661
+ and a profile-declared absolute root would sail through, aiming the window
662
+ at a state root — and so a per-project registry — its own parent cannot
663
+ see. Removing the key instead makes the child inherit the parent's own
664
+ (broken) value and fail exactly as the parent fails: whatever a child
665
+ concludes is what a clean process under the same conditions concludes, in
666
+ the error arm too. The strip governs only what froid-loop *adds* to a
667
+ child; a value already in the environment a child inherits is not
668
+ scrubbed here.
669
+ """
670
+ pinned = dict(env)
671
+ try:
672
+ pinned[envvars.STATE_DIR] = str(state_root())
673
+ except StateRootError:
674
+ pinned.pop(envvars.STATE_DIR, None)
675
+ return pinned
676
+
677
+
678
+ def state_dir_for(project: Path, run_id: str) -> Path:
679
+ """This run's control-plane directory: ``<state root>/<project key>/<run id>``.
680
+
681
+ The project key is :func:`project_tag`, reused verbatim rather than re-derived:
682
+ it already resolves the project before digesting it, so the two spellings of
683
+ one project a caller can arrive with — a symlinked path, a relative one — key
684
+ to the same directory. They must, or a run started through one spelling would
685
+ write its events where a poll through the other never looks, and the run would
686
+ wait out ``session_timeout_min`` with the completion signal sitting on disk.
687
+ Its ``resolve()`` raising on a project the OS cannot canonicalize is correct
688
+ here for the same reason: an unknowable location cannot be keyed at all, and
689
+ guessing one is the wrong-directory write the tag exists to prevent.
690
+
691
+ ``run_id`` needs no sanitizing — the id contract (see :data:`RUN_ID_RE`) is
692
+ already "a legal path segment on every platform", pinned by
693
+ :func:`is_valid_run_id`, and an id from outside is rejected there rather than
694
+ coerced here.
695
+ """
696
+ return project_state_root(project) / run_id
697
+
698
+
699
+ def events_dir_for(project: Path, run_id: str) -> Path:
700
+ """The run's hook-event channel: the directory the relay writes a session's
701
+ events into and ``SignalWatcher`` polls for them."""
702
+ return state_dir_for(project, run_id) / "events"
703
+
704
+
705
+ def config_digest_path_for(project: Path, run_id: str) -> Path:
706
+ """The run's host-exec config baseline: ``runsetup.config_digest`` as of the
707
+ last time a human started or resumed this run (#498).
708
+
709
+ Out here rather than in ``state.json`` because the baseline exists to police
710
+ the agent-writable tree, and until this move it *lived* in it: a session that
711
+ rewrote ``policy.toml`` could blank or re-stamp the field in the same breath
712
+ and the warning `resume` owes the operator never fired. The same reasoning the
713
+ events channel moved on (#494).
714
+
715
+ **What moving it buys, stated exactly.** It closes the *incidental* path: the
716
+ pin is no longer a project file, so nothing a session does in the ordinary
717
+ course of rewriting the tree can collaterally blank it — which is the case the
718
+ advisory was documented to catch. It is **not** a boundary against a
719
+ deliberate one. Sessions run with permission bypass by default — every shipped
720
+ profile's ``bypass_args``, which ``GenericAdapter.interactive_argv`` uses
721
+ unless ``[adapter] extra_args`` overrides them; that is what an unattended loop
722
+ is — and are handed ``FROID_LOOP_EVENTS_DIR``, whose parent is this directory. A
723
+ session that goes looking can *truncate* this file and the reader below answers
724
+ ``""`` — a real "no baseline" — or delete it and blank the in-tree copy
725
+ (``RunState.trusted_config_digest``, the secondary this falls back to) for the
726
+ same silence. Either way the result is indistinguishable from a run that never
727
+ had a baseline: any marker saying "this run *should* have one" would have to
728
+ live somewhere the same session cannot reach, and no such place exists at equal
729
+ privilege. Closing it needs privilege separation on the state root, not a better
730
+ hiding place — tracked in #571."""
731
+ return state_dir_for(project, run_id) / CONFIG_DIGEST_FILE
732
+
733
+
734
+ def read_trusted_config_digest(project: Path, run_id: str) -> str | None:
735
+ """This run's persisted host-exec baseline, or ``None`` when the state root
736
+ holds none for it.
737
+
738
+ ``None`` is "ask the in-tree copy", not "no pin" — the two are different
739
+ answers and the caller acts on the difference (see
740
+ ``cli._resume_paused_run``). No file here means this run's baseline is
741
+ reachable only through ``state.json``: it was paused before #498, or the
742
+ project moved and keyed its state subtree somewhere new
743
+ (:func:`project_state_root`). An *empty* file, by contrast, is a real answer
744
+ of "no baseline" and comes back as ``""``.
745
+
746
+ **Known limit: a file at this key can be stale (#572).** The key is the
747
+ project's resolved path, so a project that moves away and later returns finds
748
+ its old subtree still here — nothing can sweep it in between (FEATURES.md) —
749
+ holding the baseline blessed before it left, while the blessing it picked up
750
+ in between is the one in ``state.json``. Preferring the file means that older
751
+ pin wins for one resume, which re-stamps this key and heals it. Preferring the
752
+ fresher-looking in-tree copy is *not* the fix: it is session-writable, so it
753
+ would hand any session the silencing #498 closed. Arbitrating by sequence
754
+ number needs a counterpart the session cannot forge, and at equal privilege
755
+ there is none — the same wall as #571, reached by re-keying instead of
756
+ tampering.
757
+
758
+ Pure observation, so it degrades rather than raising: a state root this host
759
+ cannot name, or a file it cannot read, both answer ``None`` and hand the
760
+ decision to the in-tree copy. The write half raises — see
761
+ :func:`write_trusted_config_digest` — and the split is the standard one
762
+ (``platform_util.resolve_or_lexical`` states the doctrine). Degrading here
763
+ costs at most one advisory warning; a resume that *aborts* because an
764
+ advisory could not be read would be the worse failure, and the resume is
765
+ about to resolve the same state root for its events channel anyway, where
766
+ the error is owned and reported.
767
+
768
+ **Deliberately not ``read_text``**, and for the same reason the write is
769
+ ``follow_symlinks=False``: this file sits in a directory the driven session
770
+ can reach (its parent is the ``FROID_LOOP_EVENTS_DIR`` the engine exports), so
771
+ the *shape* of what is at the path has to be established before any bytes are
772
+ consumed. Degrading on a hostile path is not enough when the read itself is
773
+ the weapon:
774
+
775
+ * ``O_NONBLOCK`` + an ``S_ISREG`` check **on the descriptor**. Opening a FIFO
776
+ for reading otherwise blocks until someone writes — indefinitely — and
777
+ ``resume`` is a foreground command a human is waiting on, so a planted FIFO
778
+ wedges the terminal rather than costing a warning. The check is on the fd,
779
+ not the path, so it cannot be raced: ``fstat`` describes the object actually
780
+ opened.
781
+ * ``O_NOFOLLOW``, so the name is read rather than wherever it points.
782
+ * At most :data:`_MAX_DIGEST_BYTES`. A link to an endless source
783
+ (``/dev/zero``) reads forever otherwise, and raises ``MemoryError`` — not
784
+ the ``OSError`` this promises never to leak. The cap removes the condition
785
+ instead of absorbing it.
786
+
787
+ The POSIX-only flags degrade to 0 on win32, which has neither FIFOs at these
788
+ paths nor ``O_NOFOLLOW``; the size cap and the regular-file check carry there
789
+ on their own. This mirrors ``tui.launch._read_ctl_window`` deliberately — same
790
+ hazard, same shape, one idiom. It does **not** collapse empty to ``None`` the
791
+ way that twin does: here the two are different answers (above).
792
+
793
+ None of this makes the baseline tamper-*proof* — a session can still delete
794
+ the file, and #571 carries that. It stops a tampered path from hanging or
795
+ exhausting the orchestrator, which is a different and fixable harm."""
796
+ try:
797
+ path = config_digest_path_for(project, run_id)
798
+ except (StateRootError, OSError, RuntimeError):
799
+ return None
800
+ flags = os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0) | getattr(os, "O_NONBLOCK", 0)
801
+ flags |= getattr(os, "O_BINARY", 0) # win32: no CRLF translation on the raw fd
802
+ try:
803
+ fd = os.open(path, flags)
804
+ except OSError:
805
+ return None
806
+ try:
807
+ if not stat.S_ISREG(os.fstat(fd).st_mode):
808
+ return None
809
+ data = os.read(fd, _MAX_DIGEST_BYTES)
810
+ except OSError:
811
+ return None
812
+ finally:
813
+ os.close(fd)
814
+ try:
815
+ return data.decode("utf-8").strip()
816
+ except UnicodeDecodeError:
817
+ return None
818
+
819
+
820
+ def write_trusted_config_digest(project: Path, run_id: str, digest: str) -> None:
821
+ """Stamp ``digest`` as this run's host-exec baseline, creating the state dir.
822
+
823
+ Raises rather than degrading — a repair write, and a silently skipped stamp
824
+ is the outcome hardest to detect later: the next resume reads no file and
825
+ decides on the in-tree copy alone, which is the tree this baseline exists to
826
+ police. The caller is starting or resuming a run and is about to resolve the
827
+ very same state root for its events channel, so a root that cannot be named
828
+ or written fails that run regardless; failing here just fails it sooner,
829
+ before the pid lands.
830
+
831
+ **Call this only after the run dir exists.** Creating the state dir is what
832
+ makes this the earliest writer into it, and :func:`reconcile_orphan_state_dirs`
833
+ reads its entries *before* the live run-dir names on the strength of run dirs
834
+ being created strictly first — a state dir minted ahead of its run dir would
835
+ look like an orphan to a ``clean`` racing the launch."""
836
+ path = config_digest_path_for(project, run_id)
837
+ path.parent.mkdir(parents=True, exist_ok=True)
838
+ # Confined to the state root (#593): a machine-minted record under a root
839
+ # whose path the driven session is handed (FROID_LOOP_EVENTS_DIR names its
840
+ # sibling), so a planted link here must be replaced, never written through to
841
+ # whatever it aims at. Refusing a link at the FINAL component was not enough —
842
+ # `mkstemp(dir=...)` and `os.replace`'s destination still resolved every
843
+ # directory above by name, and the `mkdir` on the line above ACCEPTS a
844
+ # symlinked directory, so a link planted at either session-reachable component
845
+ # (`<project tag>/`, `<run id>/`) survived the setup step and redirected both
846
+ # the temp and the published stamp. `state_root()` is the one component the
847
+ # anchored walk starts from rather than checks, and it is a host fact this
848
+ # process derives — not a path any session names. The trailing newline is for
849
+ # the operator who cats the file.
850
+ atomic_write_text_confined(path, digest + "\n", confine_root=state_root())
851
+
852
+
853
+ # ---------------------------------------------------- run resolution / liveness
854
+
855
+
856
+ def run_dir_for(project: Path, run_id: str) -> Path:
857
+ return project / RUNS_DIR / run_id
858
+
859
+
860
+ def is_run(run_dir: Path) -> bool:
861
+ """A directory is a run iff it holds a state.json."""
862
+ return (run_dir / STATE_FILE).is_file()
863
+
864
+
865
+ class RunRefError(Exception):
866
+ """A run ref matched no run, or was ambiguous."""
867
+
868
+
869
+ def short_ref(run_id: str) -> str:
870
+ """The trailing hex segment — the minimal handle users type."""
871
+ return run_id.rsplit("-", 1)[-1]
872
+
873
+
874
+ def _is_path_escape(ref: str) -> bool:
875
+ """True when ``ref`` would steer ``run_dir_for``'s recomposition outside the
876
+ runs dir — it is absolute/drive-qualified, climbs with ``..``, names the runs
877
+ dir itself rather than anything inside it, or carries a path separator of
878
+ either flavour. Sub-check of the run-id charset rather than `is_valid_run_id`
879
+ itself: a run dir created by an older version (or by hand) may bear a name we
880
+ would no longer mint, and must stay addressable.
881
+
882
+ `names_tree_root` restores this site to the three-guard pairing every sibling
883
+ already spells (`policy.py`, `adapters/profile.py`, `plugins/manifest.py`); it
884
+ was the only member of the family omitting it (#480). It closes the spellings
885
+ that recompose to the runs *root* instead of a run in it. ``""`` and ``"."``
886
+ join to it exactly — measured here, both `runs / ""` and `runs / "."` *are*
887
+ the runs dir — so a `state.json` lying at that root made the exact branch
888
+ below hand `delete_run` the whole runs tree to `rmtree`. ``"..."``, ``".. "``
889
+ and ``" "`` are the Win32 half of the same rule (cited, not measurable on
890
+ POSIX): the trim of trailing periods and spaces leaves ``..`` or nothing, so
891
+ they name `.froid-loop/` or the runs dir there while both pure pathlib flavours
892
+ keep them as ordinary one-segment names.
893
+
894
+ Addressability is unharmed: skipping the exact branch only defers to partial
895
+ matching, and a legacy dir named ``"..."`` is still enumerated by
896
+ `list_run_dirs` and still matched by its own spelling.
897
+
898
+ `names_win32_alias`, the family's fourth member, is deliberately NOT applied
899
+ here — it would make a legacy run dir named ``NUL`` or ``run. `` permanently
900
+ unaddressable, which is the one thing this guard exists to prevent. Refusing
901
+ to *mint* such a name is `is_valid_run_id`'s job, and it already does it with
902
+ a `safe_segment` identity check."""
903
+ return (
904
+ is_absolute_path(ref)
905
+ or has_parent_ref(ref)
906
+ or names_tree_root(ref)
907
+ or "/" in ref
908
+ or "\\" in ref
909
+ )
910
+
911
+
912
+ def resolve_run_dir(project: Path, ref: str) -> Path:
913
+ """Full or partial run id -> its run dir. An exact id wins outright;
914
+ otherwise a partial matches when the trailing segment starts with `ref` or
915
+ the full id ends with `ref` (run ids are date-prefixed, so the tail is what
916
+ distinguishes them). Raises RunRefError on no match / ambiguity.
917
+
918
+ The exact branch recomposes a path from the raw ref, so it is skipped for any
919
+ ref that could escape the runs dir (`froid-loop delete ../../x` would otherwise
920
+ rmtree an outside directory that happens to hold a state.json). Such a ref
921
+ falls through to partial matching, which can only ever yield a name
922
+ `list_run_dirs` enumerated — and so cannot escape.
923
+
924
+ An EMPTY ref is refused outright rather than deferred: `""` is a prefix and a
925
+ suffix of every name, so partial matching reads it as a wildcard — harmlessly
926
+ ambiguous with two runs, but silently resolving the sole run of a one-run
927
+ project, which handed `froid-loop delete ""` that run. No addressability is
928
+ lost (no directory can be named `""`); every other escape spelling keeps the
929
+ partial fallback so a legacy dir named `"..."` stays matchable by its own
930
+ spelling."""
931
+ if not ref:
932
+ raise RunRefError("empty run ref: it would match every run, never name one")
933
+ if not _is_path_escape(ref):
934
+ exact = run_dir_for(project, ref)
935
+ if is_run(exact):
936
+ return exact
937
+ matches = [
938
+ d
939
+ for d in list_run_dirs(project)
940
+ if short_ref(d.name).startswith(ref) or d.name.endswith(ref)
941
+ ]
942
+ if not matches:
943
+ raise RunRefError(f"no such run: {ref}")
944
+ if len(matches) > 1:
945
+ listing = "\n".join(f" {d.name}" for d in matches)
946
+ raise RunRefError(f"ambiguous run ref {ref!r} matches {len(matches)} runs:\n{listing}")
947
+ return matches[0]
948
+
949
+
950
+ def read_pid(run_dir: Path) -> int | None:
951
+ """The recorded engine pid, or None when missing/unparseable. Reads the first
952
+ whitespace token, tolerating both the legacy pid-only file and the
953
+ ``"<pid> <identity>"`` form (see :func:`read_pid_identity`)."""
954
+ return read_pid_identity(run_dir)[0]
955
+
956
+
957
+ def read_pid_identity(run_dir: Path) -> tuple[int | None, float | None]:
958
+ """The recorded engine pid and its persisted identity, from ``<run_dir>/engine.pid``.
959
+ Thin wrapper over :func:`read_named_pid_identity` (which other pid files — the
960
+ Unity dialog probe's — reuse)."""
961
+ return read_named_pid_identity(run_dir / PID_FILE)
962
+
963
+
964
+ def read_named_pid_identity(pidfile: Path) -> tuple[int | None, float | None]:
965
+ """The pid and its persisted identity recorded in ``pidfile``. ``(None, None)``
966
+ when the file is missing or the pid is unparseable; identity ``None`` for a legacy
967
+ pid-only file (callers then degrade to a bare existence check). A malformed
968
+ second token is not legacy: it returns an impossible identity so reuse guards
969
+ fail closed. First token is the pid, an optional second token the identity float."""
970
+ try:
971
+ tokens = pidfile.read_text(encoding="utf-8").split()
972
+ except OSError:
973
+ return None, None
974
+ if not tokens:
975
+ return None, None
976
+ try:
977
+ pid = int(tokens[0])
978
+ except ValueError:
979
+ return None, None
980
+ identity: float | None = None
981
+ if len(tokens) > 1:
982
+ try:
983
+ parsed = float(tokens[1])
984
+ except ValueError:
985
+ parsed = _INVALID_PID_IDENTITY
986
+ # Only a true one-token legacy file degrades to bare existence. If an
987
+ # identity token is present but corrupt/non-finite, fail closed as not-ours.
988
+ identity = parsed if math.isfinite(parsed) else _INVALID_PID_IDENTITY
989
+ return pid, identity
990
+
991
+
992
+ def engine_alive(run_dir: Path) -> bool:
993
+ """True only when a local engine pid is provably alive **and still our engine**
994
+ (identity-checked, so a reused pid reads as dead). Mirrors :func:`liveness`
995
+ minus the tmux fallback — callers here want a definite 'is something running'
996
+ answer, and 'unknown' must not block stop/delete."""
997
+ pid, identity = read_pid_identity(run_dir)
998
+ if pid is None:
999
+ return False
1000
+ return get_process_host().alive_and_ours(pid, identity)
1001
+
1002
+
1003
+ def engine_liveness(run_dir: Path) -> str:
1004
+ """Tri-state read of the local engine: ``'alive'`` | ``'dead'`` | ``'unknown'``.
1005
+ Wraps :meth:`ProcessHost.liveness_of` so a live-but-unreadable pid (win32
1006
+ ``ERROR_ACCESS_DENIED``) reads ``'unknown'``, not a false ``'dead'``. No pid →
1007
+ ``'dead'`` (the session fallback lives in the TUI layer)."""
1008
+ pid, identity = read_pid_identity(run_dir)
1009
+ if pid is None:
1010
+ return "dead"
1011
+ return probe_liveness(pid, identity)
1012
+
1013
+
1014
+ def probe_liveness(pid: int, identity: float | None) -> str:
1015
+ """Tri-state probe of an already-read ``(pid, identity)`` — the shared body of
1016
+ :func:`engine_liveness` and :func:`liveness`, so both read the pid file once.
1017
+ A probe failure degrades to ``'unknown'``, never a false ``'dead'``."""
1018
+ host = get_process_host() # ProcessHostError (misconfig) propagates, not masked as unknown
1019
+ try:
1020
+ return host.liveness_of(pid, identity)
1021
+ except Exception:
1022
+ return "unknown"
1023
+
1024
+
1025
+ # ------------------------------------------------- run inventory / classification
1026
+
1027
+
1028
+ # Run statuses reported by `froid-loop list` and the dashboard.
1029
+ RUNNING = "running"
1030
+ PAUSED = "paused"
1031
+ FINISHED = "finished"
1032
+ STOPPED = "stopped"
1033
+ CRASHED = "crashed"
1034
+ INTERRUPTED = "interrupted"
1035
+ UNKNOWN = "unknown"
1036
+
1037
+ _StatSig = tuple[int, int, int]
1038
+
1039
+
1040
+ def _stat_sig(path: Path) -> _StatSig | None:
1041
+ try:
1042
+ st = path.stat()
1043
+ except OSError:
1044
+ return None
1045
+ # st_ino joins (mtime_ns, size): the engine rewrites state.json atomically
1046
+ # (temp + os.replace), so every write lands on a fresh inode. That catches a
1047
+ # same-size rewrite within one coarse mtime tick (e.g. WSL2 drvfs, or any fast
1048
+ # rewrite on a low-resolution mtime) that (mtime_ns, size) alone would miss and
1049
+ # serve stale from cache.
1050
+ return (st.st_mtime_ns, st.st_size, st.st_ino)
1051
+
1052
+
1053
+ def liveness(run_dir: Path) -> str:
1054
+ """'alive' | 'dead' | 'unknown' for the engine that owns run_dir.
1055
+
1056
+ engine.pid is authoritative (written at run/sweep/resume start, never
1057
+ deleted). Legacy runs without one fall back to the per-run agent session —
1058
+ but that session only exists while an agent session runs, so its absence
1059
+ proves nothing: 'unknown', never falsely dead. Pid checks are local-only;
1060
+ runs on other hosts always come back 'unknown'.
1061
+ """
1062
+ pid, identity = read_pid_identity(run_dir)
1063
+ if pid is None:
1064
+ return _session_liveness(run_dir.name)
1065
+ # Probe the pid we just read (shared body with engine_liveness) rather than
1066
+ # re-reading it, so a non-atomic pid rewrite can't split the two reads and flash
1067
+ # a false 'dead' between "pid present" here and a re-read seeing an empty file.
1068
+ try:
1069
+ return probe_liveness(pid, identity)
1070
+ except ProcessHostError:
1071
+ # A misconfigured host (bad FROID_LOOP_PROCESS_HOST) stays a hard error on
1072
+ # CLI decision paths, but the display layer must degrade, not crash: the
1073
+ # dashboard poll worker has no except and would take the whole app down.
1074
+ return "unknown"
1075
+
1076
+
1077
+ def _session_liveness(run_id: str) -> str:
1078
+ # An absent multiplexer / dead query proves nothing about a legacy run, so the
1079
+ # only positive signal is a live session; everything else is 'unknown'.
1080
+ mux = get_multiplexer()
1081
+ if not mux_usable(mux): # forced-aware, like every other observer gate
1082
+ return "unknown"
1083
+ try:
1084
+ return "alive" if mux.has_session(session_name(run_id)) else "unknown"
1085
+ except (OSError, MultiplexerError):
1086
+ # The seam raises MultiplexerError (not OSError) on a backend failure; a
1087
+ # dead query proves nothing about a legacy run, so degrade to 'unknown'
1088
+ # rather than crashing the TUI poll.
1089
+ return "unknown"
1090
+
1091
+
1092
+ def _classify(finished: bool, paused: bool, stopped: bool, crashed: bool, run_dir: Path) -> str:
1093
+ if finished:
1094
+ return FINISHED
1095
+ if paused:
1096
+ return PAUSED
1097
+ # a deliberate stop leaves a dead pid — check it before liveness so it does
1098
+ # not read as INTERRUPTED (a crash).
1099
+ if stopped:
1100
+ return STOPPED
1101
+ # a recorded crash leaves a dead pid too — surface it as a distinct CRASHED
1102
+ # before liveness, where it would otherwise read as a generic INTERRUPTED.
1103
+ if crashed:
1104
+ return CRASHED
1105
+ live = liveness(run_dir)
1106
+ if live == "alive":
1107
+ return RUNNING
1108
+ if live == "dead":
1109
+ return INTERRUPTED
1110
+ return UNKNOWN
1111
+
1112
+
1113
+ @dataclass(frozen=True)
1114
+ class RunInfo:
1115
+ run_id: str
1116
+ run_dir: Path
1117
+ run_type: str
1118
+ started_at: str
1119
+ status: str
1120
+ paused_stage: str = "" # RunState.paused_stage when PAUSED, else ""; drives the badge
1121
+ stopping: bool = False # a graceful stop is pending (control file present) while RUNNING
1122
+
1123
+
1124
+ # state.json path -> (stat sig, header fields tuple)
1125
+ _HeaderFields = tuple[str, str, bool, bool, bool, bool, str]
1126
+ _header_cache: dict[Path, tuple[_StatSig, _HeaderFields]] = {}
1127
+
1128
+
1129
+ def discover_runs(project: Path) -> list[RunInfo]:
1130
+ """One RunInfo per run dir, oldest first; [] when the runs dir is missing.
1131
+
1132
+ Parses only the state.json header fields (cached on stat); a state file
1133
+ that fails to parse yields status 'unknown' rather than crashing — it is
1134
+ transient, the engine writes atomically.
1135
+ """
1136
+ out: list[RunInfo] = []
1137
+ for run_dir in list_run_dirs(project):
1138
+ state_path = run_dir / STATE_FILE
1139
+ sig = _stat_sig(state_path)
1140
+ cached = _header_cache.get(state_path)
1141
+ if sig is not None and cached is not None and cached[0] == sig:
1142
+ run_type, started_at, finished, paused, stopped, crashed, paused_stage = cached[1]
1143
+ else:
1144
+ try:
1145
+ doc = json.loads(state_path.read_text(encoding="utf-8"))
1146
+ run_type = str(doc.get("run_type", "story"))
1147
+ started_at = str(doc.get("started_at", ""))
1148
+ finished = bool(doc.get("finished", False))
1149
+ paused = doc.get("paused_reason") is not None
1150
+ stopped = bool(doc.get("stopped", False))
1151
+ crashed = bool(doc.get("crashed", False))
1152
+ paused_stage = str(doc.get("paused_stage") or "")
1153
+ except (OSError, json.JSONDecodeError):
1154
+ out.append(RunInfo(run_dir.name, run_dir, "?", "", UNKNOWN))
1155
+ continue
1156
+ if sig is not None:
1157
+ _header_cache[state_path] = (
1158
+ sig,
1159
+ (run_type, started_at, finished, paused, stopped, crashed, paused_stage),
1160
+ )
1161
+ status = _classify(finished, paused, stopped, crashed, run_dir)
1162
+ # paused_stage is advisory: only meaningful while the run is actually PAUSED
1163
+ # (a resumed run keeps the last stage in state until it re-pauses/finishes).
1164
+ stage = paused_stage if status == PAUSED else ""
1165
+ # A pending graceful stop is the control file's presence, but only while an
1166
+ # engine is still around to honor it — RUNNING or UNKNOWN (an unverifiable
1167
+ # pid still consumes the file). The engine discards the file at the stop
1168
+ # boundary, so a lingering file on an already-concluded run is not "stopping":
1169
+ # STOPPED/FINISHED/CRASHED classify before liveness, so they never read UNKNOWN.
1170
+ stopping = status in (RUNNING, UNKNOWN) and (run_dir / STOP_REQUEST_FILE).is_file()
1171
+ out.append(RunInfo(run_dir.name, run_dir, run_type, started_at, status, stage, stopping))
1172
+ return out
1173
+
1174
+
1175
+ # ----------------------------------------------------------- stop / delete / archive
1176
+
1177
+
1178
+ def kill_session(run_id: str, mux: TerminalMultiplexer | None = None) -> None:
1179
+ """Kill a run's agent session (froid-loop-<id>); a no-op when it is already
1180
+ gone or the multiplexer is unavailable.
1181
+
1182
+ Also a no-op for an id that **aliases a control session**
1183
+ (:func:`run_id_aliases_control_session` — ``ctl`` or ``ctl-<16 hex>``,
1184
+ case-folded): the only session such a name can address is the control
1185
+ plane, every parked window of every run in it. Unreachable through
1186
+ minting (validation refuses the shape) but reachable through what an
1187
+ **older release persisted**: a run dir named ``ctl`` that `stop`,
1188
+ `delete` or a resume's stale-session sweep replays as a kill target.
1189
+ This chokepoint keeps those read paths safe — and usable as the
1190
+ operator's way out of such a run — without each caller re-deriving the
1191
+ rule.
1192
+
1193
+ The narrow test, not the mint's broad reservation, deliberately: a
1194
+ historical ``ctl-foo`` run DOES own an agent session of its own
1195
+ (``froid-loop-ctl-foo``, distinct from every control session and killed
1196
+ exactly — the seam sends ``=``-exact tmux targets, and psmux resolves
1197
+ exact port files), and skipping its kill stranded it: the prune already
1198
+ could not reach it, so nothing could. Scope, stated: the kill addresses
1199
+ the registry THIS process addresses — a pre-upgrade session left in
1200
+ psmux's old default registry is not reachable from here (measured), and
1201
+ deliberately so: a by-name kill in a shared registry without tag proof
1202
+ could take another project's same-named session (run ids are unique per
1203
+ project only). The legacy sweep in :func:`prune_sessions`, which does
1204
+ demand the tag, is the path that reaches it."""
1205
+ if run_id_aliases_control_session(run_id):
1206
+ return
1207
+ (mux or get_multiplexer()).kill_session(session_name(run_id))
1208
+
1209
+
1210
+ CTL_SESSION = "froid-loop-ctl"
1211
+ _SESSION_PREFIX = "froid-loop-"
1212
+
1213
+
1214
+ def ctl_session_for(project: Path, mux: TerminalMultiplexer | None = None) -> str:
1215
+ """The control-session name this project's launches and lookups share.
1216
+
1217
+ On a transport with no registry namespace (tmux) it is the fixed
1218
+ :data:`CTL_SESSION`, machine-shared as it has always been. On a namespacing
1219
+ transport (psmux) the name carries the registry's identity — a 16-hex
1220
+ digest of the derived registry root — because the two scopes genuinely
1221
+ differ: the session lives *per registry*, but psmux's duplicate-server
1222
+ guard is a mutex keyed on the session name alone, across every registry
1223
+ in the **login session** (``Local\\psmux-session-{name}`` over
1224
+ ``port_file_base()`` — the ``Local\\`` kernel-object namespace is
1225
+ per-login-session, not machine-global; ``server/mod.rs:853`` /
1226
+ ``platform.rs:346`` / ``types.rs:1345``, source-read at v3.3.8 —
1227
+ ``PSMUX_DATA_DIR`` never enters it). A fixed name therefore admits ONE
1228
+ control session across every registry a desktop session can reach,
1229
+ and the second project's create is rejected as a duplicate server — its
1230
+ TUI launch fails instead of minting its own session (measured: a second
1231
+ registry answers ``new-session`` rc 1 for the fixed name while the first
1232
+ registry's server lives, and rc 0 for a per-registry name).
1233
+
1234
+ The digest is over ``mux_registry_root(project)`` **resolved**: the name
1235
+ must be unique per *physical* registry, and the resolved path is that
1236
+ registry's identity — (project, state root), both axes; ``project_tag``
1237
+ alone would recreate the collision for one project under two state roots.
1238
+ Resolved rather than as spelled because two spellings of one state root
1239
+ (``C:\\work\\state`` vs ``C:\\work\\alias\\..\\state``) reach **one**
1240
+ registry — Windows resolves both to the same files, and psmux keeps the
1241
+ spelling only while constructing those paths (``src/paths.rs:79``,
1242
+ source-read at v3.3.8; convergence measured) — so an as-spelled digest
1243
+ minted two control sessions inside one registry, each blind to the other's
1244
+ parked windows: the split-control-plane failure again, one level up. Same
1245
+ rule ``project_tag`` already states: resolve *before* digesting.
1246
+
1247
+ …and then ``os.path.normcase``, because ``resolve()`` can only return the
1248
+ filesystem's stored case for a path that **exists**, and the registry
1249
+ root usually does not yet exist at the moment the name is needed (psmux
1250
+ ``create_dir_all``\\s it at first spawn). Two case spellings of a
1251
+ not-yet-created state root resolve to two strings, digest to two names —
1252
+ and then land in ONE physical registry, because NTFS folds case when
1253
+ psmux opens the ``.port`` files (measured). ``normcase`` folds exactly
1254
+ where the filesystem does: it lowercases on Windows and is the identity
1255
+ on POSIX, where case is significant and two case spellings ARE two
1256
+ registries — folding there would merge genuinely distinct roots.
1257
+ Ceiling, named: ``normcase`` lowercases with ``str.lower``, which can
1258
+ disagree with NTFS's own fold table for a few non-ASCII case pairs; a
1259
+ state root spelled in two such casings of the same non-ASCII name stays
1260
+ split, as it is for every other digest of an operator-supplied path.
1261
+
1262
+ The degrade arm (namespaced transport, underivable state root) answers
1263
+ the fixed name: that arm runs on the transport's shared default registry,
1264
+ where a shared session scoped by per-window project tags is the correct,
1265
+ tmux-shaped semantic — and where a pre-#537 legacy ctl session under the
1266
+ fixed name may exist to be reused rather than collided with.
1267
+ """
1268
+ mux = mux or get_multiplexer()
1269
+ if not mux.has_registry_namespace():
1270
+ return CTL_SESSION
1271
+ try:
1272
+ scope = os.path.normcase(str(mux_registry_root(project).resolve()))
1273
+ except (StateRootError, OSError, RuntimeError):
1274
+ return CTL_SESSION
1275
+ return f"{CTL_SESSION}-{hashlib.sha256(os.fsencode(scope)).hexdigest()[:16]}"
1276
+
1277
+
1278
+ def is_ctl_session_name(name: str) -> bool:
1279
+ """Whether ``name`` is a control session's name — the fixed
1280
+ :data:`CTL_SESSION`, or ``froid-loop-ctl-<16 hex>``, the ONE suffix shape
1281
+ :func:`ctl_session_for` can mint.
1282
+
1283
+ The shape predicate exists because several readers ask "is this A control
1284
+ session" without a project in hand: the agent-session parser must exclude
1285
+ ctl sessions (``froid-loop-ctl-<16hex>`` would otherwise parse as run id
1286
+ ``ctl-<16hex>``, which ``RUN_ID_RE`` admits), the legacy-leftovers reader
1287
+ names a surviving ctl session in a registry this process did not derive,
1288
+ and ``in_ctl_session`` classifies whatever session this process woke up
1289
+ inside.
1290
+
1291
+ Exactly the mintable shapes, no wider: an arbitrary suffix
1292
+ (``froid-loop-ctl-foo``) is NOT a control session — it is the agent
1293
+ session of a run an older release accepted as ``--run-id ctl-foo``, and
1294
+ reading it as a control session made it unreachable by ``stop`` and the
1295
+ prune both. No agent session of OURS can match this predicate:
1296
+ :func:`is_valid_run_id` refuses every ctl-shaped id at the mint (broad —
1297
+ :func:`is_reserved_run_id`), so a matching name is either genuinely a
1298
+ control session or hand-made to look like one — and the hand-made
1299
+ 16-hex-suffixed case stays unprunable, the leak direction."""
1300
+ if name == CTL_SESSION:
1301
+ return True
1302
+ suffix = name.removeprefix(CTL_SESSION + "-")
1303
+ return suffix != name and len(suffix) == 16 and all(c in "0123456789abcdef" for c in suffix)
1304
+
1305
+
1306
+ # tmux user option stamping a session/window with the project it belongs to, so
1307
+ # a prune in one project never touches another project's live runs. See
1308
+ # prunable_sessions and tui.launch.
1309
+ PROJECT_OPTION = "@froid_project"
1310
+
1311
+
1312
+ def project_tag(project: Path) -> str:
1313
+ """Canonical project identity used by both tag writers and prune readers. The
1314
+ single source of normalization: both sides must route through this so symlinks
1315
+ and relative paths can't make a project look foreign to its own sessions.
1316
+
1317
+ Hashing the resolved path makes every value safe by construction, on both
1318
+ transports a tag has to cross. It clears psmux's control line (#419), whose
1319
+ gate refuses any value the CLI->server hop would mangle — a UNC share whose
1320
+ name holds a space is refused verbatim, and that refusal left the session
1321
+ untagged, which is weak ownership twice over. It equally clears the listing
1322
+ round trip (#518): a hex digest holds nothing `str.splitlines()` breaks on,
1323
+ no tab, and no byte outside ASCII, so it can neither split a row nor fail the
1324
+ backends' strict decode.
1325
+
1326
+ That subsumes the conditional percent-encoding this function briefly applied.
1327
+ Encoding answered only the listing half, so a path the listing could carry but
1328
+ the control line could not — the spaced UNC above — still went untagged. The
1329
+ compatibility objection encoding was shaped around, that rewriting every tag
1330
+ strands the ones already stored on live sessions and windows, is answered on
1331
+ the read side instead, by `accepted_tags`.
1332
+
1333
+ 16 hex characters are ample for one machine's project population.
1334
+ """
1335
+ return hashlib.sha256(os.fsencode(str(project.resolve()))).hexdigest()[:16]
1336
+
1337
+
1338
+ def accepted_tags(project: Path) -> frozenset[str]:
1339
+ """Current digest plus the legacy resolved-path tag accepted during pruning.
1340
+
1341
+ The legacy member is read-only compatibility for sessions and ctl windows that
1342
+ survive an upgrade; remove it once no path-tagged multiplexer state can remain.
1343
+ Returns the whole set rather than answering per tag so a read site resolves the
1344
+ project once per prune instead of once per session.
1345
+
1346
+ The two shapes cannot collide into false ownership: a legacy tag is an absolute
1347
+ path, so it always holds a separator, while a digest is bare 16-hex.
1348
+
1349
+ Deliberately two members and not three — a tag spelled with the `%enc%` prefix,
1350
+ from the window when this module encoded rather than hashed, is not accepted.
1351
+ Only a path the listing could not carry was ever spelled that way (one holding
1352
+ a line separator, or a byte invalid in the filesystem encoding), and that
1353
+ spelling never reached a release. An unaccepted tag reads as foreign, which
1354
+ skips the session rather than pruning it, so the edge is fail-safe and clears
1355
+ itself on the next tag write.
1356
+ """
1357
+ return frozenset({project_tag(project), str(project.resolve())})
1358
+
1359
+
1360
+ def lock_path_for(data_path: Path) -> Path:
1361
+ """The advisory-lock sidecar for a mutable data file:
1362
+ ``<state root>/locks/<sha256(resolved path)[:16]>-<basename>.lock``.
1363
+
1364
+ Out of the repository, deliberately, and NOT the ``<file>.lock`` sibling the
1365
+ obvious reading of #286 asks for. The deferred-work ledger is a *tracked*
1366
+ file by design, and both :func:`verify.commit_story` and
1367
+ :func:`verify.finalize_commit` stage with ``git add -A``: a lock beside it
1368
+ would be swept into the engine's own commits, and the git-add shield that
1369
+ would otherwise hide it covers linked worktrees only. Under the state root
1370
+ the sidecar is never git-visible at all, so no exclusion machinery has to be
1371
+ kept correct for it.
1372
+
1373
+ Keyed on the **resolved** path so the identity of the lock is the identity of
1374
+ the file rather than of the spelling used to reach it: a symlinked and a
1375
+ direct path to one ledger rendezvous on one lock (without which the two
1376
+ spellings would exclude nobody), two worktrees' in-tree ledgers are different
1377
+ files and correctly get independent locks, and several projects pointed at a
1378
+ shared external artifact dir land on one lock, which is where the real
1379
+ contention is. The basename is appended for debuggability only — a human
1380
+ reading ``ls`` of the locks dir should see which file a sidecar guards — and
1381
+ carries no meaning for exclusion, which rides the digest.
1382
+
1383
+ Pure: no directory is created here, because
1384
+ :func:`~froid_loop.platform_util.file_lock` mkdirs the parent when it opens
1385
+ the lock. May raise :class:`StateRootError` when the environment names no
1386
+ usable state root (see :func:`state_root`); the caller fails rather than
1387
+ silently locking somewhere else.
1388
+ """
1389
+ resolved = data_path.resolve()
1390
+ digest = hashlib.sha256(os.fsencode(str(resolved))).hexdigest()[:16]
1391
+ return state_root() / "locks" / f"{digest}-{resolved.name}.lock"
1392
+
1393
+
1394
+ def mux_sessions() -> list[str]:
1395
+ """All live session names, or [] when the multiplexer is missing, no server
1396
+ is running, or the query fails."""
1397
+ return get_multiplexer().list_sessions()
1398
+
1399
+
1400
+ def session_project_tags() -> dict[str, str]:
1401
+ """Map each live session name to its PROJECT_OPTION value ("" when unset).
1402
+ Same missing-multiplexer/no-server guards as mux_sessions()."""
1403
+ return get_multiplexer().session_options(PROJECT_OPTION)
1404
+
1405
+
1406
+ def _agent_run_id(session: str) -> str | None:
1407
+ """The run id behind a ``froid-loop-<id>`` agent session name, or ``None`` when
1408
+ the name is not one: the control session, a foreign session, or a mangled name
1409
+ whose id could not be replayed as a path segment — never let one steer a
1410
+ run-dir path. Shared so the prune partition and
1411
+ :func:`legacy_registry_leftovers` cannot drift on what counts as ours.
1412
+
1413
+ The id question is :func:`is_parsable_run_id`, deliberately NOT
1414
+ :func:`is_valid_run_id` — the parse side must accept ids the mint refuses.
1415
+ That predicate owns the reasoning, and the ctl-window sweep in
1416
+ ``tui.launch`` asks the same one."""
1417
+ if not session.startswith(_SESSION_PREFIX):
1418
+ return None
1419
+ run_id = session[len(_SESSION_PREFIX) :]
1420
+ return run_id if is_parsable_run_id(run_id) else None
1421
+
1422
+
1423
+ def prunable_sessions(
1424
+ project: Path, mux: TerminalMultiplexer | None = None, *, require_tag: bool = False
1425
+ ) -> tuple[list[str], list[str], set[str]]:
1426
+ """Partition the froid-loop-<id> agent sessions into (prunable, live) run ids,
1427
+ plus the subset of prunable ids whose engine liveness read 'unknown'
1428
+ (unverifiable pid). Unknown never blocks cleanup — those sessions stay
1429
+ prunable — but frontends surface a warning for them.
1430
+
1431
+ A control session (:func:`is_ctl_session_name` — the fixed name or a
1432
+ per-registry one) is never a candidate. Pruning is scoped
1433
+ to `project` via the PROJECT_OPTION tag set at session creation:
1434
+
1435
+ - tag proves this project (see accepted_tags): ours — prunable unless a
1436
+ provably-alive engine pid is running (covers finished/stopped/crashed *and*
1437
+ orphans whose run dir was deleted, since engine_liveness reads 'dead' with
1438
+ no pid).
1439
+ - tag is another project: skipped — never touched.
1440
+ - tag empty (untagged session): can't prove ownership, so fall back to the run
1441
+ dir — prunable only when the dir exists under this project and is dead;
1442
+ skipped when the dir is absent. Reachable when the tag write failed, when
1443
+ the option read degrades (session_options reads unset as "no answer", never
1444
+ as proof nothing was written), or on a session predating a working tag
1445
+ write — e.g. psmux path tags refused before the digest.
1446
+
1447
+ ``require_tag`` drops that last arm: an untagged session is skipped outright
1448
+ rather than falling back to the run dir. Set for a **shared** registry — the
1449
+ legacy psmux root every project's pre-upgrade sessions sit in together (see
1450
+ :func:`prune_sessions`). The fallback proves ownership from
1451
+ ``run_dir_for(project, run_id)``, and a run id is only unique *within* one
1452
+ project (``--run-id`` is caller-supplied), so in a shared registry a dead run
1453
+ dir here is not evidence about a session over there: project A holding a dead
1454
+ `shared-id` would claim project B's live, untagged `froid-loop-shared-id` and
1455
+ kill it. In a per-project registry the same fallback is sound because the
1456
+ registry itself proves ownership, which is why the flag is off by default and
1457
+ the primary pass keeps the reach it always had. What the flag leaves behind is
1458
+ reported by :func:`legacy_registry_leftovers`.
1459
+ """
1460
+ # `mux` bypasses the module-level readers rather than widening them: those
1461
+ # two are the seam every other caller (and every test) reaches the process-wide
1462
+ # backend through, and a bound instance is this function's business alone.
1463
+ tags = mux.session_options(PROJECT_OPTION) if mux is not None else session_project_tags()
1464
+ mine = accepted_tags(project)
1465
+ prunable: list[str] = []
1466
+ live: list[str] = []
1467
+ unknown: set[str] = set()
1468
+ names = mux.list_sessions() if mux is not None else mux_sessions()
1469
+ for name in names:
1470
+ run_id = _agent_run_id(name)
1471
+ if run_id is None:
1472
+ continue
1473
+ run_dir = run_dir_for(project, run_id)
1474
+ tag = tags.get(name, "")
1475
+ if tag:
1476
+ if tag not in mine:
1477
+ continue # another project's session
1478
+ elif require_tag or not is_run(run_dir):
1479
+ continue # ownership unprovable: no tag, and no run dir here to stand in
1480
+ liveness = engine_liveness(run_dir)
1481
+ if liveness == "alive":
1482
+ live.append(run_id)
1483
+ continue
1484
+ prunable.append(run_id)
1485
+ if liveness == "unknown":
1486
+ unknown.add(run_id)
1487
+ return prunable, live, unknown
1488
+
1489
+
1490
+ def _registry_proves_ownership(project: Path) -> bool:
1491
+ """True when the registry the *primary* prune pass addresses is one froid-loop
1492
+ derived for this project — which is what makes
1493
+ :func:`prunable_sessions`' untagged run-dir fallback evidence rather than a
1494
+ guess.
1495
+
1496
+ That fallback claims an untagged ``froid-loop-<id>`` session when this project
1497
+ holds a dead run dir of the same id. Run ids are only unique *within* a
1498
+ project (``--run-id`` is caller-supplied), so the claim is sound exactly when
1499
+ the registry itself already restricts what can be listed to this project's
1500
+ sessions. In a registry shared with other projects — or with the operator —
1501
+ it is not, and the same reasoning that put ``require_tag=True`` on the legacy
1502
+ pass applies here.
1503
+
1504
+ The primary registry is not always ours. :func:`export_psmux_registry_root`
1505
+ degrades to ``None`` on an underivable state root and leaves whatever ambient
1506
+ ``PSMUX_DATA_DIR`` it found in force, and psmux honours any absolute value
1507
+ (``src/paths.rs``, source-read at v3.3.8) — so on that arm every verb,
1508
+ including the kill, addresses the operator's own registry while this project's
1509
+ run dirs go on looking like ownership.
1510
+
1511
+ ``registry_root()`` answering ``None`` covers two cases, and they get
1512
+ **opposite** answers — conflating them was a defect, not caution. A backend
1513
+ with no registry namespace at all (tmux: one server for the machine,
1514
+ ``has_registry_namespace()`` False) keeps the reach it had before
1515
+ per-project registries existed: the listing there is exactly what it always
1516
+ was, and narrowing it would be a regression dressed as caution. A backend
1517
+ that DOES namespace and has no root in force (psmux with ``PSMUX_DATA_DIR``
1518
+ unset — the export degraded on an underivable state root and there was no
1519
+ ambient value either) is running on its transport's own **default**
1520
+ registry — shared with every other project and with the operator
1521
+ (``<home>\\.psmux``, the home being ``USERPROFILE`` when set, else the
1522
+ profile API, else ``HOMEDRIVE``+``HOMEPATH``, else ``HOME`` —
1523
+ ``src/paths.rs`` ``home_dir``, source-read at v3.3.8) — which proves nothing about
1524
+ ownership, exactly as an absolute ambient value naming a foreign registry
1525
+ proves nothing. Both shared cases make the tag mandatory.
1526
+
1527
+ A backend that cannot be asked answers ``False``: the safe direction is to
1528
+ demand the tag, which leaves a session standing rather than killing one on
1529
+ evidence that may not hold.
1530
+ """
1531
+ try:
1532
+ mux = get_multiplexer()
1533
+ root = mux.registry_root()
1534
+ if root is None:
1535
+ # No namespace (tmux): historical reach. A namespace with no root
1536
+ # in force is the transport's shared default registry: demand the tag.
1537
+ return not mux.has_registry_namespace()
1538
+ except MultiplexerError:
1539
+ return False
1540
+ try:
1541
+ return root == str(mux_registry_root(project))
1542
+ except (StateRootError, OSError, RuntimeError):
1543
+ return False
1544
+
1545
+
1546
+ def prune_sessions(
1547
+ project: Path, *, dry_run: bool = False
1548
+ ) -> tuple[list[str], list[str], set[str]]:
1549
+ """Kill every prunable froid-loop-<id> session (see prunable_sessions);
1550
+ returns (killed, live, unknown): the run ids that were (or, with dry_run,
1551
+ would be) killed, the live ids skipped, and the killed subset whose engine
1552
+ liveness read 'unknown'. All three come from the same partition sample, so
1553
+ frontend messaging built from them always describes the performed actions.
1554
+
1555
+ Runs once per registry: the one this process is pointed at, then each legacy
1556
+ registry the backend still admits (:func:`_legacy_registries`). Sessions
1557
+ froid-loop created before it took a per-project psmux root are addressable
1558
+ only from the second pass, and without it cleanup would report a clean sweep
1559
+ while their servers ran on. The passes are unioned rather than concatenated —
1560
+ a run id can only be in one registry, but a backend answering the same
1561
+ registry twice must not make one kill look like two.
1562
+
1563
+ Ownership is judged per pass by the same :func:`prunable_sessions` partition,
1564
+ so a legacy registry buys no extra reach: another project's sessions and the
1565
+ operator's own psmux sessions are skipped there exactly as they are here.
1566
+
1567
+ The legacy pass always runs with ``require_tag=True``, and the primary pass
1568
+ runs with it whenever the registry it addresses is not one froid-loop derived
1569
+ for this project (:func:`_registry_proves_ownership`). Both are the same rule:
1570
+ :func:`prunable_sessions`' untagged run-dir fallback is evidence only where
1571
+ the registry has already restricted the listing to this project. A legacy
1572
+ registry is shared by every project by definition; the primary one is shared
1573
+ whenever the derivation failed — an ambient ``PSMUX_DATA_DIR`` left in
1574
+ force, or nothing in force at all, where a namespacing backend runs on its
1575
+ own shared default registry. What that strictness leaves standing in a
1576
+ legacy registry is reported
1577
+ by :func:`legacy_registry_leftovers`, which the cleanup frontends print: a
1578
+ sweep that silently declines to migrate something is the same silence this
1579
+ whole change exists to remove."""
1580
+ prunable, live, unknown = prunable_sessions(
1581
+ project, require_tag=not _registry_proves_ownership(project)
1582
+ )
1583
+ if not dry_run:
1584
+ for run_id in prunable:
1585
+ kill_session(run_id)
1586
+ for legacy in _legacy_registries():
1587
+ extra, extra_live, extra_unknown = prunable_sessions(project, legacy, require_tag=True)
1588
+ if not dry_run:
1589
+ for run_id in extra:
1590
+ kill_session(run_id, legacy)
1591
+ prunable += [i for i in extra if i not in prunable]
1592
+ live += [i for i in extra_live if i not in live]
1593
+ unknown |= extra_unknown
1594
+ return prunable, live, unknown
1595
+
1596
+
1597
+ #: How a frontend names psmux's OWN default registry, the one root
1598
+ #: :meth:`~.adapters.multiplexer.TerminalMultiplexer.registry_root` deliberately
1599
+ #: answers ``None`` for (respelling its home cascade in Python is a second thing
1600
+ #: to keep in sync). Lives here so both frontends say it the same way.
1601
+ DEFAULT_REGISTRY_LABEL = "the multiplexer's own default registry"
1602
+
1603
+
1604
+ def legacy_registry_leftovers(
1605
+ project: Path, *, announced: Iterable[str] = ()
1606
+ ) -> dict[str, list[str]]:
1607
+ """Session names a legacy registry **still holds** after :func:`prune_sessions`
1608
+ ran — the migration's honest remainder, for the cleanup frontends to print.
1609
+ ``{}`` when there is no legacy registry, when they hold nothing, or when
1610
+ every listing fails.
1611
+
1612
+ **Grouped by registry, and that is load-bearing.** There is more than one
1613
+ legacy registry now (:meth:`~.adapters.psmux_backend.PsmuxMultiplexer.legacy_registries`
1614
+ — psmux's default, and the root this process displaced), so a flat list
1615
+ cannot say where to go look: a message built from one would either name a
1616
+ registry the leftovers are not in, or name every registry the sweep
1617
+ addressed including the ones that contributed nothing. The operator's next
1618
+ action is to open that registry, so the answer has to be per registry. Keys
1619
+ are :meth:`registry_root`'s answer, or :data:`DEFAULT_REGISTRY_LABEL` where
1620
+ that is ``None``; a registry holding nothing is absent rather than empty, so
1621
+ a caller can print the keys without checking.
1622
+
1623
+ **Presence, not a second opinion.** Called after the sweep, this lists what is
1624
+ actually there; a session the sweep killed is simply gone from the listing.
1625
+ That is the whole judgement for anything tagged as ours, and it is deliberately
1626
+ *not* a re-run of the partition: re-judging liveness would open a race the
1627
+ reader cannot see the far side of. A run alive during the prune (correctly
1628
+ left, and reported ``live``) can exit before the reader looks; a resampled
1629
+ partition would then call it ``prunable``, and it would fall out of both the
1630
+ live arm and the untagged fallback — stranded and unreported, with no kill ever
1631
+ attempted. Presence has no such gap: the session is standing, so it is named.
1632
+
1633
+ What that covers, in one rule:
1634
+
1635
+ - **Untagged** ``froid-loop-<id>`` sessions. The legacy pass runs with
1636
+ ``require_tag=True`` (:func:`prunable_sessions`), so an untagged session there
1637
+ is skipped rather than claimed by a run dir that proves nothing in a shared
1638
+ registry.
1639
+ - **Ours, still standing.** Tagged this project's, and the sweep did not remove
1640
+ it — because it was live, because it exited mid-sweep, or because
1641
+ ``kill_session`` (best-effort and silent by contract) did not land. All three
1642
+ leave the same fact behind: a session of ours in a registry ordinary attach
1643
+ and cleanup no longer address. Naming it needs no cause, which is why this
1644
+ also closes the failed-kill case ``cleanup --json``'s ``sessions.removed``
1645
+ documents as an *attempted* kill.
1646
+ - **A surviving control session.** The prune never touches a ctl-named
1647
+ session (:func:`is_ctl_session_name`), and its parked windows are not swept
1648
+ in a legacy registry either — the ctl-window scan runs against the primary
1649
+ backend only. The shape question is asked through *that registry's*
1650
+ :meth:`~.adapters.multiplexer.TerminalMultiplexer.session_name_key`, never a
1651
+ constant fold: on a case-folding store ``froid-loop-CTL-<hex>`` IS the
1652
+ control session and goes unnamed without it, while on an exact one it is a
1653
+ distinct session froid-loop cannot have minted — naming it there would send
1654
+ the operator after somebody else's.
1655
+
1656
+ Another project's tagged sessions never appear: the sweep skipping them is the
1657
+ correct outcome, not a remainder.
1658
+
1659
+ ``announced`` is the one thing presence alone cannot judge: on a **dry run**
1660
+ nothing was killed, so every session the preview just announced as a would-kill
1661
+ is still standing and would be named here as if the sweep had declined it. The
1662
+ caller passes the run ids it printed — :func:`prune_sessions`' own return — and
1663
+ they are excluded.
1664
+
1665
+ **Excluded only where THIS registry's own pass could have announced it**, which
1666
+ is the tagged-ours arm and only it. :func:`prune_sessions` unions the ids of
1667
+ every pass, the *primary* registry's included, so the flat set says no more
1668
+ than "some registry would kill this id" — while a legacy pass runs with
1669
+ ``require_tag=True`` and therefore cannot claim an untagged session at all.
1670
+ Applied to the untagged arm the set hid exactly the remainder this listing
1671
+ exists for: a dead ``froid-loop-X`` the primary pass plans to kill, an untagged
1672
+ ``froid-loop-X`` over here that the real cleanup leaves and reports, and a
1673
+ preview of that same cleanup that does not mention it.
1674
+
1675
+ Inside the tagged arm the flat set is exact, so no per-registry plan has to be
1676
+ threaded down here. Liveness is read from ``run_dir_for(project, run_id)`` —
1677
+ one directory per (project, id), whatever registry the session sits in — so an
1678
+ id the primary pass judged dead the legacy pass judges dead too: if the same
1679
+ id is standing here under a tag proving ours, this pass announced it as well
1680
+ and the union merely collapsed the two.
1681
+
1682
+ Passed in rather than re-derived, and that is the whole point of the parameter.
1683
+ An earlier revision re-ran the partition here to rediscover the plan, which is
1684
+ a *second sample*: a tagged legacy run seen alive by the first (so printed as
1685
+ live, never announced) can exit before this call, land in the second sample's
1686
+ prunable arm, and be excluded from a listing it should have headed — a session
1687
+ dropped from the preview outright, not merely mentioned twice. Consuming what
1688
+ the preview actually printed cannot disagree with it.
1689
+
1690
+ On a real cleanup the caller passes nothing: there, a killed session is gone
1691
+ from the listing by presence, and one whose kill did not land must be named.
1692
+
1693
+ Deliberately its own listing rather than a fourth arm on
1694
+ :func:`prune_sessions`. That tuple is read by two frontends and projected into
1695
+ the schema-versioned ``cleanup --json`` document; widening it is a contract
1696
+ change and ~30 call sites, against one extra pair of psmux calls against a
1697
+ registry that answers "no server" instantly on any machine that never ran the
1698
+ pre-registry build.
1699
+
1700
+ Names, not run ids: the ctl session has no run id, and the operator is going to
1701
+ paste these into a ``psmux`` target — under the registry this maps them to,
1702
+ which is the other half of what makes them pasteable.
1703
+ """
1704
+ grouped: dict[str, list[str]] = {}
1705
+ mine = accepted_tags(project)
1706
+ # Run ids, so names. `prune_sessions` unions its passes, so an id it reports
1707
+ # names at most one session anywhere — the same collapse that makes its own
1708
+ # "killed" count one per id.
1709
+ excluded = {session_name(run_id) for run_id in announced}
1710
+ for legacy in _legacy_registries():
1711
+ try:
1712
+ names = legacy.list_sessions()
1713
+ tags = legacy.session_options(PROJECT_OPTION) if names else {}
1714
+ except MultiplexerError:
1715
+ continue # observation degrades; the sweep's own report still stands
1716
+ here: list[str] = []
1717
+ for name in names:
1718
+ if is_ctl_session_name(legacy.session_name_key(name)):
1719
+ # A legacy registry holds the pre-#537 fixed name; the shape
1720
+ # predicate also names any per-registry-named stray. Asked
1721
+ # through THIS registry's own comparison key, never a constant
1722
+ # fold: whether `froid-loop-CTL-<hex>` denotes the control
1723
+ # session is the transport's answer to give.
1724
+ here.append(name)
1725
+ continue
1726
+ if _agent_run_id(name) is None:
1727
+ continue # not a froid-loop agent session at all
1728
+ tag = tags.get(name, "")
1729
+ if tag and tag not in mine:
1730
+ continue # another project's session
1731
+ if tag and name in excluded:
1732
+ continue # a would-kill of this registry's own pass (dry run)
1733
+ here.append(name)
1734
+ if here:
1735
+ # `registry_root()` is a diagnostic and never raises (seam contract).
1736
+ # Two admitted registries could in principle answer the same label —
1737
+ # a displaced root that spells the default is swept twice — so the
1738
+ # rows are merged rather than overwritten.
1739
+ label = legacy.registry_root() or DEFAULT_REGISTRY_LABEL
1740
+ grouped[label] = sorted(set(grouped.get(label, []) + here))
1741
+ return grouped
1742
+
1743
+
1744
+ def _legacy_registries() -> list[TerminalMultiplexer]:
1745
+ """Backends bound to registries this project's sessions may predate, or []
1746
+ (see :meth:`~.multiplexer.TerminalMultiplexer.legacy_registries`, which owns
1747
+ the concept and every backend's answer).
1748
+
1749
+ Degrades to [] rather than raising: a backend that cannot even be selected
1750
+ has no legacy registry to offer, and a cleanup that already swept the primary
1751
+ registry must report that work rather than die on the migration pass."""
1752
+ try:
1753
+ return list(get_multiplexer().legacy_registries())
1754
+ except MultiplexerError:
1755
+ return []
1756
+
1757
+
1758
+ # The run dir of the OUTERMOST engine in this call stack (#319). A nested auto-sweep
1759
+ # runs synchronously in its parent's thread but mints its own run id and dir, so its
1760
+ # adapters would poll a control file no operator ever writes to: `froid-loop stop
1761
+ # <parent-id>` lodges in the parent's dir. This carries the owning run dir down to
1762
+ # them. A ContextVar, mirroring `engine._run_depth`, because the nesting it tracks is
1763
+ # same-thread by construction; set once by the outermost `Engine.run()` and reset by
1764
+ # token, so a later top-level run in the same process is never poisoned.
1765
+ _owner_run_dir: contextvars.ContextVar[Path | None] = contextvars.ContextVar(
1766
+ "froid_loop_owner_run_dir", default=None
1767
+ )
1768
+
1769
+
1770
+ def set_owner_run_dir(run_dir: Path) -> contextvars.Token[Path | None]:
1771
+ """Claim ``run_dir`` as the owning run for this call stack. Returns the token the
1772
+ caller must hand to :func:`reset_owner_run_dir` from a ``finally``."""
1773
+ return _owner_run_dir.set(run_dir)
1774
+
1775
+
1776
+ def reset_owner_run_dir(token: contextvars.Token[Path | None]) -> None:
1777
+ """Release the claim made by :func:`set_owner_run_dir`."""
1778
+ _owner_run_dir.reset(token)
1779
+
1780
+
1781
+ def owner_run_dir() -> Path | None:
1782
+ """The outermost engine's run dir, or None outside any run — which is what a
1783
+ standalone adapter (tests, probes) reads, so callers fall back to their own."""
1784
+ return _owner_run_dir.get()
1785
+
1786
+
1787
+ def graceful_stop_requested(run_dir: Path) -> bool:
1788
+ """True when *some* stop request is pending for this run — either mode. A bare
1789
+ existence read of the control file, never raising and deliberately never parsing.
1790
+
1791
+ Every consumer wants exactly that existence question, not the mode: the
1792
+ ``stopping`` projection and the TUI badge (a run with a hard request lodged is
1793
+ stopping too), the ``--graceful`` idempotency check (a lodged hard request means
1794
+ a *stronger* stop already stands — "already-pending" is the right answer), the
1795
+ stories done-checkpoint skip, and auto-sweep suppression. Only ``status``'s
1796
+ ``graceful_stop_pending`` field is mode-exact; it calls
1797
+ :func:`read_stop_request_mode` instead."""
1798
+ return (run_dir / STOP_REQUEST_FILE).is_file()
1799
+
1800
+
1801
+ def read_stop_request_mode(run_dir: Path) -> str | None:
1802
+ """The mode of this run's pending stop request: ``"hard"``, ``"graceful"``, or
1803
+ ``None`` when none is pending.
1804
+
1805
+ ``None`` means *absent*, and only absent — it is returned for
1806
+ ``FileNotFoundError`` alone. Everything else about a file that is *present*
1807
+ reads ``"graceful"``: a modeless body (every pre-#319 writer and test fixture
1808
+ wrote one — this is the back-compat pin), unparseable or non-object JSON, and a
1809
+ transient read failure such as the win32 sharing violation a concurrent
1810
+ ``atomic_replace`` raises mid-write.
1811
+
1812
+ Leaning graceful on every ambiguity is load-bearing, not defensive habit. A
1813
+ misread graceful costs at most one more item before the run stops; a spurious
1814
+ ``"hard"`` would abort a live session — so a torn read must never be able to
1815
+ produce one."""
1816
+ return _stop_request_mode_of(run_dir / STOP_REQUEST_FILE)
1817
+
1818
+
1819
+ def _stop_request_mode_of(path: Path) -> str | None:
1820
+ """The parse half of :func:`read_stop_request_mode`, split out so
1821
+ :func:`consume_stop_request` can answer for the file it *took* rather than for
1822
+ whatever currently answers to the channel name."""
1823
+ try:
1824
+ raw = path.read_text(encoding="utf-8")
1825
+ except FileNotFoundError:
1826
+ return None
1827
+ except (OSError, ValueError):
1828
+ # present but unreadable this tick (sharing violation, undecodable bytes) —
1829
+ # answer for the file we know is there, never escalate on a failed read.
1830
+ return "graceful"
1831
+ try:
1832
+ body = json.loads(raw)
1833
+ except ValueError:
1834
+ return "graceful"
1835
+ if isinstance(body, dict) and body.get("mode") == "hard":
1836
+ return "hard"
1837
+ return "graceful"
1838
+
1839
+
1840
+ def _project_of_run_dir(run_dir: Path) -> Path:
1841
+ """The project root a run directory hangs under, for confining writes into it.
1842
+
1843
+ Derived rather than passed because the stop-request channel is addressed by
1844
+ run directory alone: `stop_run` resolves a run reference and never holds the
1845
+ project separately. :func:`run_dir_for` is the only builder of these paths
1846
+ and spells them ``project / RUNS_DIR / run_id``, so the root sits exactly
1847
+ ``len(RUNS_DIR.parts)`` levels above the run's own directory — the arithmetic
1848
+ tracks `RUNS_DIR` rather than hard-coding 2, so moving the runs tree moves
1849
+ this with it.
1850
+
1851
+ A path too shallow to have that ancestor is not one this module built.
1852
+ Refusing with :class:`UnconfinedWriteError` rather than letting `parents`
1853
+ raise `IndexError` is the load-bearing part: `stop_run` degrades on `OSError`
1854
+ so that a failed lodge still signals the run, and an `IndexError` there would
1855
+ abort the stop before it ever signalled."""
1856
+ depth = len(RUNS_DIR.parts)
1857
+ parents = run_dir.parents
1858
+ if len(parents) <= depth:
1859
+ raise UnconfinedWriteError(f"{run_dir} is not shaped like a run directory")
1860
+ return parents[depth]
1861
+
1862
+
1863
+ def _write_stop_request(run_dir: Path, mode: str) -> None:
1864
+ """Lodge a stop request of ``mode`` on the control-file channel, written
1865
+ atomically so a concurrent engine read never sees a partial body.
1866
+
1867
+ The atomic replace *is* the supersede: writing ``"hard"`` over a pending
1868
+ ``"graceful"`` escalates the request in one step, with no window in which
1869
+ nothing is pending for the engine to find.
1870
+
1871
+ That is the only direction this function arbitrates, and the only one it may:
1872
+ ``stop_run`` shares it and its escalation must stay unconditional. The channel is
1873
+ otherwise last-writer-wins, so the *reverse* — a graceful write landing on a
1874
+ pending hard request and downgrading it — is refused by a different writer
1875
+ entirely: :func:`_create_stop_request`, which lodges the graceful mode with
1876
+ ``O_CREAT | O_EXCL`` so "is one pending?" and "lodge mine" are a single atomic
1877
+ step. Splitting the two directions across two functions is what lets this one
1878
+ stay an unconditional replace.
1879
+
1880
+ Goes through :func:`platform_util.atomic_write_text` rather than a hand-rolled
1881
+ ``tmp + atomic_replace``, for the reason ``operatoractions`` was migrated under
1882
+ #379: this is the one control file with genuinely *concurrent* writers — two
1883
+ ``stop`` invocations against the same run, in either mode — and a fixed ``.tmp``
1884
+ sibling is exactly what two writers of the same key collide on. Interleaved,
1885
+ both stage over one name and the loser's ``os.replace`` raises
1886
+ ``FileNotFoundError`` after the winner's consumed it; on the hard path that
1887
+ would abort ``stop_run`` *before* it ever signals. A ``mkstemp`` temp per writer
1888
+ removes the collision: the last replace wins and neither writer errors.
1889
+
1890
+ Refusing a link at the control file preserved what the bare ``os.replace``
1891
+ did — it never dereferenced this destination — and matches what the file is:
1892
+ machine-minted control state under a run dir a driven session can reach. The
1893
+ write is now confined to the project root (#593), because that refusal
1894
+ covered only the final component: every directory above it was still looked
1895
+ up by name, so a link planted at ``.froid-loop/`` — or at ``runs/``, or at the
1896
+ run's own directory — aimed both the temp and the publish wherever it
1897
+ pointed. The file still lands at ``mkstemp``'s ``0600`` instead of
1898
+ ``0644 & ~umask``, since no-follow never inherited a mode either; nothing
1899
+ reads it cross-user.
1900
+
1901
+ No ``require_writable_target``: this is not an operator-curated file but a
1902
+ channel two ``stop`` invocations race on, and its whole contract above is
1903
+ that the stronger request always lands."""
1904
+ body = json.dumps({"requested_at": time.strftime("%Y-%m-%dT%H:%M:%S"), "mode": mode})
1905
+ atomic_write_text_confined(
1906
+ run_dir / STOP_REQUEST_FILE, body, confine_root=_project_of_run_dir(run_dir)
1907
+ )
1908
+
1909
+
1910
+ def _create_stop_request(run_dir: Path) -> bool:
1911
+ """Lodge a *graceful* request only if none is pending; False when one already is.
1912
+
1913
+ ``O_CREAT | O_EXCL`` is the arbitration. It makes "is a request pending?" and
1914
+ "lodge mine" one atomic step against the destination name, so a hard request
1915
+ landing at any instant either already exists — we refuse, leaving it standing —
1916
+ or replaces what we wrote, which is escalation, the direction
1917
+ :func:`_write_stop_request` owns. A re-read immediately before an unconditional
1918
+ replace could only ever *narrow* that window (~1.3ms on a journalling
1919
+ filesystem, where the fsync dominates); this closes it.
1920
+
1921
+ Graceful-ONLY by construction, and that is what makes the non-atomic body safe.
1922
+ The bytes are written *into* the created file rather than replaced in, so a
1923
+ concurrent reader can catch it empty — and :func:`read_stop_request_mode`
1924
+ answers ``"graceful"`` for a present-but-unparseable body, which is the very
1925
+ mode being written. The invariant that matters is untouched: a torn read must
1926
+ never produce ``"hard"``, so a hard writer must keep the atomic replace.
1927
+
1928
+ Refuses a planted symlink rather than following it — ``O_EXCL`` never
1929
+ dereferences — which is stricter than the ``follow_symlinks=False`` replace it
1930
+ replaces. That refusal covers only the FINAL component, though, so the create
1931
+ goes through :func:`platform_util.create_exclusive_confined` (#593): a link
1932
+ planted at ``.froid-loop/``, ``runs/`` or the run's own directory was still
1933
+ resolved by name and aimed the request outside the project, exactly the hole
1934
+ the confined :func:`_write_stop_request` next door already closed. The
1935
+ anchored create keeps the exclusive arbitration this function is built on;
1936
+ an unreachable parent raises ``UnconfinedWriteError``.
1937
+
1938
+ A failed write is deliberately NOT rolled back, and that is load-bearing rather
1939
+ than sloppy. ``unlink`` resolves a *name*, not the inode this call created, so a
1940
+ rollback here would delete whatever occupies the path at that moment — including
1941
+ a ``"hard"`` request a concurrent ``stop`` escalated onto it while this write was
1942
+ in flight. That is a ``hard -> absent`` drop, the one descent the mode lattice
1943
+ :func:`consume_stop_request` documents must never happen, and on native Windows
1944
+ it would silently withdraw the only channel that can stop the engine. Guarding it
1945
+ is not available: an "unlink only if still my inode" step does not exist as one
1946
+ atomic operation, and both check-then-unlink shapes measure *worse* than no guard
1947
+ at all — the check moves the decision earlier and the destructive act later by
1948
+ its own cost, shifting the window rather than narrowing it (inode compare 1.39x,
1949
+ mode compare 2.30x, over a rendezvous-synchronised escalation sweep on btrfs).
1950
+
1951
+ What a failed write leaves behind is a short or empty body, which
1952
+ :func:`read_stop_request_mode` reads as ``"graceful"`` — exactly the mode this
1953
+ function was asked to lodge, for the one caller (``stop --graceful``) that an
1954
+ operator drove. It does not wedge the channel: a later graceful ask answers
1955
+ "already-pending", a later *hard* stop supersedes it unconditionally, and
1956
+ ``stop --cancel-graceful`` or ``resume`` withdraws it. Leaving a graceful request
1957
+ standing is the bounded direction this channel already leans on everywhere else."""
1958
+ body = json.dumps({"requested_at": time.strftime("%Y-%m-%dT%H:%M:%S"), "mode": "graceful"})
1959
+ path = run_dir / STOP_REQUEST_FILE
1960
+ try:
1961
+ fd = create_exclusive_confined(path, confine_root=_project_of_run_dir(run_dir))
1962
+ except FileExistsError:
1963
+ return False # a request is already pending — a planted link included
1964
+ with os.fdopen(fd, "w", encoding="utf-8") as fh:
1965
+ fh.write(body)
1966
+ return True
1967
+
1968
+
1969
+ def clear_graceful_stop(run_dir: Path) -> bool:
1970
+ """Consume a pending stop request of *either* mode, returning True iff one was
1971
+ present and removed. Never raises: the engine calls this the moment it honors a
1972
+ request, a resume calls it to discard a stale one, and stop_run calls it on the
1973
+ paths where nothing is left alive to read what it lodged — a missing file
1974
+ (already consumed) or an unremovable one must not wedge any of them. Uses the
1975
+ same win32 sharing-violation retry the atomic write pairs with."""
1976
+ try:
1977
+ retrying_unlink(run_dir / STOP_REQUEST_FILE)
1978
+ except OSError:
1979
+ # FileNotFoundError (nothing pending) or a genuine removal failure — either
1980
+ # way nothing was discarded, and the caller must not see an exception.
1981
+ return False
1982
+ return True
1983
+
1984
+
1985
+ def consume_stop_request(run_dir: Path) -> str | None:
1986
+ """Take the pending request off the channel and answer the mode of the very file
1987
+ removed, or ``None`` when none was pending.
1988
+
1989
+ The reader-side counterpart of :func:`_create_stop_request`'s
1990
+ ``O_CREAT | O_EXCL``: the rename *is* the consume, so "what mode is pending?"
1991
+ and "take it" cannot disagree. A read followed by an unlink can, and the gap is
1992
+ not academic — a concurrent ``stop`` escalating to ``"hard"`` in between is
1993
+ deleted unread while the caller routes on the stale ``"graceful"`` it already
1994
+ holds.
1995
+
1996
+ Only that direction can lose anything, because the mode lattice is monotone:
1997
+ the graceful writer refuses to overwrite an existing request and the hard writer
1998
+ only ever writes ``"hard"``, so ``absent < graceful < hard`` until consumed. A
1999
+ stale ``"hard"`` read is therefore always still true; a stale ``"graceful"`` may
2000
+ not be.
2001
+
2002
+ Monotone requires that no writer *descends* either, which is why
2003
+ :func:`_create_stop_request` has no rollback on a failed write: an ``unlink``
2004
+ keyed on the path rather than the inode it created is a ``hard -> absent`` drop,
2005
+ and it would put a second way to lose a hard request in a *writer* — leaving the
2006
+ three read-then-unlink sites in ``engine.py`` that rely on this argument resting
2007
+ on something untrue.
2008
+
2009
+ Re-reading the mode immediately before the unlink does NOT fix this, and is a
2010
+ trap worth naming: measured over 4000 injected races it made the loss *more*
2011
+ likely, not less (164 -> 929 swallowed), because the extra read lengthens the
2012
+ interval an escalation has to land in. Narrowing a window is not closing it —
2013
+ only one atomic step is.
2014
+
2015
+ A hard request lodged *after* the take is a new request against a run already
2016
+ stopping. It stays at the canonical name for ``run()``'s finally to discard and
2017
+ journal as ``stop-request-discarded`` — a record, not a silent loss."""
2018
+ src = run_dir / STOP_REQUEST_FILE
2019
+ taken = run_dir / (STOP_REQUEST_FILE + ".consumed")
2020
+ try:
2021
+ atomic_replace(src, taken)
2022
+ except FileNotFoundError:
2023
+ return None
2024
+ except OSError:
2025
+ # Could not take it (read-only dir, a sharing violation past its retries).
2026
+ # Leave it on the channel and answer from the canonical name: the next
2027
+ # boundary re-asks, which is strictly better than losing the request.
2028
+ return read_stop_request_mode(run_dir)
2029
+ try:
2030
+ return _stop_request_mode_of(taken)
2031
+ finally:
2032
+ with contextlib.suppress(OSError):
2033
+ retrying_unlink(taken)
2034
+
2035
+
2036
+ def request_graceful_stop(run_dir: Path) -> str:
2037
+ """Ask a live run to stop gracefully: finish the in-flight item (story ->
2038
+ dev/review/commit, or a sweep bundle through commit) cleanly, then finalize and
2039
+ stop — resumable, unlike the hard stop :func:`stop_run` delivers.
2040
+
2041
+ Delivery is the :data:`STOP_REQUEST_FILE` control file, written atomically by
2042
+ :func:`_write_stop_request` so a concurrent engine read never sees a partial file.
2043
+ Never signals the process and never writes ``journal.jsonl`` (engine-owned
2044
+ single-writer). Returns a status token for the caller to message on:
2045
+
2046
+ - ``"requested"`` — file written; a provably-live engine will honor it.
2047
+ - ``"already-pending"`` — a request was already on disk; left untouched so its
2048
+ original ``requested_at`` stands (idempotent — a second ask is a no-op). The
2049
+ token is mode-blind, and the pending request is not necessarily graceful: a
2050
+ *hard* one sits there at rest whenever a `stop` could not prove the engine
2051
+ dead, and one can also land while this call is in flight. A stronger stop
2052
+ stands either way and must not be downgraded — so callers message this token
2053
+ as a *stop request*, never as a graceful one (#319).
2054
+ - ``"requested-unverifiable"`` — file written, but engine liveness read
2055
+ ``'unknown'`` (e.g. a win32 access-denied pid): the request stands and fires
2056
+ if an engine is in fact running; the caller warns that it can't confirm.
2057
+
2058
+ Raises :class:`GracefulStopError` when the run has already finished (nothing to
2059
+ stop) or its engine is provably dead (no consumer — ``resume`` is the tool).
2060
+ """
2061
+ state = load_state(run_dir)
2062
+ if state.finished:
2063
+ raise GracefulStopError(f"run {run_dir.name} has already finished — nothing to stop")
2064
+ if graceful_stop_requested(run_dir):
2065
+ return "already-pending" # keep the original request's timestamp
2066
+ liveness = engine_liveness(run_dir)
2067
+ if liveness == "dead":
2068
+ raise GracefulStopError(
2069
+ f"run {run_dir.name} has no live engine — a graceful stop request would "
2070
+ f"never be consumed; use `froid-loop resume {run_dir.name}` to continue it"
2071
+ )
2072
+ # The write IS the check. The existence test at the top of this function is
2073
+ # separated from here by a pid-file read, a liveness probe and (formerly) a
2074
+ # mkstemp and an fsync — measured at ~1.3ms median on btrfs, wide enough for a
2075
+ # concurrent `stop` to lodge `"hard"` in between — and the channel is
2076
+ # last-writer-wins, so an unconditional replace here would silently *downgrade*
2077
+ # it and cost the abort the operator asked for. A re-read just before the replace
2078
+ # narrows that window; a create-if-absent removes it, because there is no longer
2079
+ # a gap between deciding and writing. "already-pending" is the same answer the
2080
+ # check at the top gives, and the right one either way: a lodged hard request is
2081
+ # a *stronger* stop already standing. Two concurrent *graceful* asks resolve the
2082
+ # same way, which is the documented idempotency — the first one's timestamp
2083
+ # stands. The escalation direction is untouched and stays unconditional.
2084
+ if not _create_stop_request(run_dir):
2085
+ return "already-pending"
2086
+ return "requested" if liveness == "alive" else "requested-unverifiable"
2087
+
2088
+
2089
+ def stop_run(run_dir: Path) -> bool:
2090
+ """Stop a live run. Returns False if it was already finished.
2091
+
2092
+ The request is delivered two ways at once, and the engine wins whichever race
2093
+ it can: a ``mode: hard`` :data:`STOP_REQUEST_FILE` is lodged *first*, then the
2094
+ engine is signalled. SIGTERM is the POSIX fast path — the handler stops the run
2095
+ within the tick. The file is what makes the stop work where the signal cannot
2096
+ land: a native-Windows engine never receives an inter-process SIGTERM, so before
2097
+ #319 every Windows stop burned the full grace window into a blind force-kill.
2098
+ Now the engine reads the file at its next item boundary, or mid-session in the
2099
+ adapter wait loop, and performs its own teardown either way.
2100
+
2101
+ That ordering is the whole point: lodging before signalling means the engine can
2102
+ never exit the signal path having missed a request that was only written after.
2103
+
2104
+ Either way the engine stays the single writer of `stopped` (it marks the run,
2105
+ kills its in-flight agent window, and exits). Falls back to an external kill +
2106
+ mark when there is no live engine pid, it is a legacy run, or it does not exit
2107
+ in time. A wedged engine that ignores both channels past the grace window is
2108
+ force-killed — but only while we can still prove the pid is the same process we
2109
+ signalled (a pid-reuse guard); otherwise we raise StopRunError rather than risk
2110
+ killing an unrelated process.
2111
+
2112
+ The lodged file is consumed by whoever settles the run: the engine when it
2113
+ honors the request, or this function on the paths where nothing is left alive to
2114
+ read it. Both exceptions to that turn on the same question — did we ever *prove*
2115
+ the engine dead? Where we did not, the file stays lodged, because it is then the
2116
+ only channel that can still stop it: the StopRunError refusal below (we decline
2117
+ to force-kill an unverifiable pid), and the ``engine_may_live`` paths where the
2118
+ signal or the kill was refused outright rather than racing us to exit.
2119
+
2120
+ **Registry scope, stated because it is easy to read past.** The stop
2121
+ itself is registry-independent: both channels address the engine *process* —
2122
+ the request file lands in the run directory, the signal on the pid recorded
2123
+ there — and a run directory is per (project, run id), not per registry. So a
2124
+ pre-upgrade run living in a legacy psmux registry stops, and a still-live
2125
+ engine tears down its own window under the registry it was launched with.
2126
+ What is scoped is the backstop below: :func:`kill_session` addresses the
2127
+ registry THIS process exported, so an agent session an already-dead engine
2128
+ leaked in a legacy registry is not reached from here and the run is marked
2129
+ stopped with that session standing.
2130
+
2131
+ Deliberately not widened, and for the reason ``kill_session``'s own docstring
2132
+ gives: a by-name kill in a registry shared with other projects, without tag
2133
+ proof, could take a neighbour's same-named session — run ids are unique per
2134
+ project only. Both legacy registries are shared in exactly that sense. The
2135
+ displaced one is no exception: it is the *ambient* ``PSMUX_DATA_DIR`` this
2136
+ process found (:func:`~.adapters.psmux_backend.note_displaced_registry`), so
2137
+ a profile that exports one exports it into every project's shell and every
2138
+ one of them kept its pre-upgrade sessions there. That is why the legacy pass
2139
+ of :func:`prune_sessions` demands the tag in both, and it is the path that
2140
+ reaches such a session — ``froid-loop cleanup``, with
2141
+ :func:`legacy_registry_leftovers` naming whatever the tag rule leaves and the
2142
+ registry it is in.
2143
+ """
2144
+ state = load_state(run_dir)
2145
+ if state.finished:
2146
+ return False
2147
+
2148
+ # Lodge the hard request before signalling. The atomic replace also supersedes a
2149
+ # pending *graceful* request in the same step: the operator escalated past it, and
2150
+ # a stronger request must never leave a window where nothing at all is pending.
2151
+ #
2152
+ # Degrade rather than abort when the lodge fails (read-only run dir, ENOSPC — and
2153
+ # the run's own session logs tee into this very directory, so a run can fill the
2154
+ # disk that then blocks stopping it). The doctrine's unit is the *repair*, not the
2155
+ # syscall: this stop is "delivered two ways at once" per the docstring above, so
2156
+ # failing the whole thing because one of two redundant channels failed would leave
2157
+ # a run alive that the pre-#319 signal path could still have killed. Keep the
2158
+ # signal, and stay loud where it actually matters — see the refusal branch below.
2159
+ try:
2160
+ _write_stop_request(run_dir, "hard")
2161
+ lodged = True
2162
+ except OSError:
2163
+ lodged = False
2164
+
2165
+ host = get_process_host()
2166
+ pid, identity = read_pid_identity(run_dir) # identity recorded at run start, not sampled now
2167
+ if pid is not None and identity is not None and not host.alive_and_ours(pid, identity):
2168
+ # the pid we recorded is already gone, or was reused by an unrelated
2169
+ # process before stop_run ran — never signal a stranger; mark stopped below.
2170
+ pid = None
2171
+ # Whether this call ever proved the engine dead. Only a confirmed death licenses
2172
+ # the fallback below to discard the request we lodged: while the engine may still
2173
+ # be running, that file is the one channel left that can stop it (on native
2174
+ # Windows it is the *only* one), so retracting it would throw away the very
2175
+ # repair #319 exists to deliver.
2176
+ engine_may_live = False
2177
+ if pid is not None:
2178
+ try:
2179
+ host.terminate(pid)
2180
+ except ProcessLookupError:
2181
+ pid = None # provably gone — the fallback's discard is correct
2182
+ except (PermissionError, OSError):
2183
+ # We could not signal it and it was `alive_and_ours` a moment ago, so it
2184
+ # may well still be running (an EPERM mismatch, or a win32 taskkill that
2185
+ # errored). Skip the wait — there is nothing to wait for — but keep the
2186
+ # request lodged so the engine can still stop itself off the file.
2187
+ engine_may_live = True
2188
+ pid = None
2189
+ if pid is not None:
2190
+ deadline = time.monotonic() + _STOP_WAIT_S
2191
+ while time.monotonic() < deadline:
2192
+ if not host.is_alive(pid):
2193
+ break # exited
2194
+ time.sleep(_STOP_POLL_S)
2195
+ if host.is_alive(pid):
2196
+ # still wedged past the grace window — escalate to a force-kill, but
2197
+ # only if this is provably the same process we signalled (never SIGKILL
2198
+ # a pid the kernel may have recycled to an unrelated process). For a
2199
+ # legacy pid file (no persisted identity) fall back to a stop-time
2200
+ # sample so a pre-upgrade run can still be force-killed — today's
2201
+ # behavior, carrying the same late-sample reuse window it always had.
2202
+ guard = identity if identity is not None else host.identity(pid)
2203
+ if guard is not None and host.identity(pid) == guard:
2204
+ try:
2205
+ host.force_kill(pid)
2206
+ except ProcessLookupError:
2207
+ pass # raced us to exit — that's the outcome we wanted
2208
+ except (PermissionError, OSError):
2209
+ # Unlike ESRCH above, this is the opposite news: the process is
2210
+ # there and we were refused. Keep the request lodged.
2211
+ engine_may_live = True
2212
+ else:
2213
+ # A kill that returned cleanly is not a death certificate — on
2214
+ # win32 `force_kill` shells `taskkill /F /T` with `check=False`,
2215
+ # so a refused kill raises nothing at all, and win32 is the
2216
+ # platform this whole channel exists for. Confirm rather than
2217
+ # infer, since the answer decides whether we discard the request.
2218
+ # Let it settle first: a delivered SIGKILL is immediate but the
2219
+ # pid can linger a moment before it is reaped, and reading that
2220
+ # as "still alive" would strand the file on the ordinary
2221
+ # wedged-engine path.
2222
+ confirm_deadline = time.monotonic() + _KILL_CONFIRM_S
2223
+ while host.is_alive(pid) and time.monotonic() < confirm_deadline:
2224
+ time.sleep(_STOP_POLL_S)
2225
+ engine_may_live = host.is_alive(pid)
2226
+ else:
2227
+ # Refusing to kill leaves the hard request lodged on purpose: if that
2228
+ # pid *is* still our engine, the file is the only channel left that
2229
+ # can stop it, and discarding it here would retract a request the
2230
+ # operator made while we decline to enforce it ourselves.
2231
+ #
2232
+ # That reasoning only holds while the lodge succeeded. If it did not,
2233
+ # nothing at all is pending and we are declining to force-kill on top
2234
+ # of that — the operator must be told, or they are left believing a
2235
+ # request is in flight that was never written.
2236
+ if lodged:
2237
+ raise StopRunError(
2238
+ f"run {run_dir.name}: engine pid {pid} honored neither the "
2239
+ "lodged stop request nor SIGTERM, and its identity can no "
2240
+ "longer be verified; refusing to force-kill a possibly-reused "
2241
+ "pid"
2242
+ )
2243
+ raise StopRunError(
2244
+ f"run {run_dir.name}: the stop request could not be written to "
2245
+ f"the run directory and engine pid {pid} did not honor SIGTERM; "
2246
+ "its identity can no longer be verified, so it will not be "
2247
+ "force-killed. No stop is pending — free space in the run "
2248
+ "directory and retry, or stop the process yourself"
2249
+ )
2250
+ # the engine clears its agent window itself, but kill the session as a backstop
2251
+ # in case it died before tearing it down. Ahead of everything below, because both
2252
+ # exits from here need it — an engine that honored the stop and died before
2253
+ # tearing its window down leaks the session just as surely as one we killed.
2254
+ # This is the one registry-scoped step of the stop (see the docstring): it
2255
+ # addresses the registry this process exported, and `cleanup`'s legacy pass is
2256
+ # what reaches a session left in an older one.
2257
+ kill_session(run_dir.name)
2258
+ state = load_state(run_dir)
2259
+ if state.stopped:
2260
+ # The engine honored the stop and is gone, and its own `run-stop` already
2261
+ # stands in the journal. Stamping `fallback=True` on top would describe an
2262
+ # engine that did its own teardown as one that had to be stopped from
2263
+ # outside. This check deliberately sits out here rather than inside the
2264
+ # `pid is not None` arm it used to live in: every path that clears `pid`
2265
+ # early — a pid that is no longer ours, a `terminate` that raced the exit
2266
+ # and got `ProcessLookupError`, a refusal that could not verify it — skipped
2267
+ # it and fell straight through to the append. The plainest case needs no race
2268
+ # at all: `stop` on a run a previous `stop` already stopped (`stopped` is set,
2269
+ # `finished` is not, so the guard at the top does not fire).
2270
+ #
2271
+ # It normally consumes the file on the way out; clear it belt-and-braces so a
2272
+ # run that is later resumed can never find our request still lodged and
2273
+ # re-stop at its first item. Safe on the `engine_may_live` paths too: a
2274
+ # written `stopped` *is* the engine reporting it honored the request, so
2275
+ # there is no live consumer left to strand.
2276
+ clear_graceful_stop(run_dir)
2277
+ return True
2278
+
2279
+ # Neither channel was delivered: nothing is lodged, and we never proved the engine
2280
+ # dead. This is the one outcome `stop` must not report as success — the operator is
2281
+ # left believing a request is in flight that was never written, while an engine we
2282
+ # could not signal keeps mutating the project. The pid-reuse guard above already
2283
+ # refuses for its own path; these are its siblings, and the only reason they stayed
2284
+ # quiet is that they clear `pid` and skip that block. Not a regression — on the
2285
+ # merge-base this was the state of *every* refused signal, because `stop_run` cleared
2286
+ # the request as its first statement — but the earlier decision to report success
2287
+ # rested on the request being retained, which is exactly what did not happen here.
2288
+ #
2289
+ # Placement is load-bearing, twice over. It sits *after* the session backstop
2290
+ # because refusing to report a stop is no reason to leak the window, and *after* the
2291
+ # `state.stopped` return because a run the engine already honored must not be
2292
+ # reported as a failure. Journal the attempt before raising: the `run-stop` append
2293
+ # below is skipped, and an unrecorded stop attempt is its own trap.
2294
+ if engine_may_live and not lodged:
2295
+ Journal(run_dir).append("run-stop-undelivered", pid=pid)
2296
+ raise StopRunError(
2297
+ f"run {run_dir.name}: the stop request could not be written to the run "
2298
+ "directory and the engine could not be proved dead, so no stop is pending. "
2299
+ "Its agent session was killed as a backstop. Free space in the run directory "
2300
+ "and retry, or stop the process yourself"
2301
+ )
2302
+
2303
+ # Fallback: no live engine (or it never confirmed). Mark it stopped here. Discard
2304
+ # the request first — nothing is left alive to consume it, and a file outliving
2305
+ # the run it asked to stop is a trap for the next resume.
2306
+ #
2307
+ # Unless we never actually proved that. Where the engine may still be running,
2308
+ # the request stays lodged and the stop is genuinely still in flight: the engine
2309
+ # honors the file at its next poll and writes `stopped` itself. Discarding it here
2310
+ # would leave a live engine with no channel left while we report the run stopped —
2311
+ # the stale-request trap above is the lesser of the two, and it only bites a run
2312
+ # that is later resumed, which this one cannot be until that engine exits.
2313
+ if not engine_may_live:
2314
+ clear_graceful_stop(run_dir)
2315
+ state.stopped = True
2316
+ save_state(run_dir, state)
2317
+ Journal(run_dir).append("run-stop", pid=pid, fallback=True)
2318
+ return True
2319
+
2320
+
2321
+ def live_session_may_be_ours(project: Path, run_id: str) -> bool:
2322
+ """True when a live ``froid-loop-<id>`` session exists that this project cannot
2323
+ prove belongs to another one — the precondition of the removal guard below.
2324
+
2325
+ Ownership is read exactly as :func:`prunable_sessions` reads it. A tag outside
2326
+ :func:`accepted_tags` proves the session foreign, and a *tagged* session carries
2327
+ its own ownership proof, so it does not need this project's run dir at all:
2328
+ answering False there keeps the guard off a removal that provably strands
2329
+ nothing. Untagged, or tagged as ours, answers True — neither can be ruled out
2330
+ as depending on this run dir, and only the untagged case is load-bearing.
2331
+
2332
+ An observation, so it degrades rather than raising, and each read degrades in
2333
+ its own direction. A listing that cannot answer reads as "no session": that is
2334
+ already what the bundled backend returns for a missing multiplexer, a dead
2335
+ server or a failed query, and a guard that varied by backend would be worse
2336
+ than no guard. A tag that cannot be read is *not* proof the session is foreign,
2337
+ so it reads as untagged and the refusal stands — by then the listing has
2338
+ already established that a session is live.
2339
+
2340
+ Both reads are caught explicitly because the seam permits a raise: only
2341
+ `pipe_pane` and `kill_session` are contractually best-effort, so an
2342
+ out-of-tree backend raises :class:`MultiplexerError` here where the bundled
2343
+ one returns empty (docs/adapter-authoring-guide.md). The listing is checked
2344
+ first, so the tag query only runs on a name collision.
2345
+
2346
+ A stronger shape was built and withdrawn: a proof discipline (block unless
2347
+ the transport *proves* the session absent) fell to four consecutive reviews,
2348
+ each refuting its newest proof source — the transports genuinely offer none.
2349
+ psmux's registry is advisory and self-healing (its server re-creates a
2350
+ reaped port file on a 5 s tick, source-read at v3.3.8), a binary's PATH
2351
+ presence is per-process while the server is not, and the listing is
2352
+ load-sensitive; so a "proof of absence" either wedges every removal behind
2353
+ `--force` or quietly accepts a refutable proof. The degrade above is the
2354
+ guard's owner's documented trade, kept deliberately; the measured cost of
2355
+ the unobservable-multiplexer window is filed for that owner to revisit
2356
+ rather than overturned here.
2357
+
2358
+ Two registry-root-era additions on that unchanged contract:
2359
+
2360
+ **The control-alias discount.** An id whose session name is one of THE
2361
+ control session's own names — the fixed :data:`CTL_SESSION`, or this
2362
+ project's :func:`ctl_session_for` — answers False before any transport
2363
+ read: that session is the control plane's, never claimed through a run
2364
+ dir, so its liveness is not evidence about the run, and blocking removal
2365
+ on it wedged exactly the recovery (`froid-loop delete ctl`) the resume
2366
+ refusal points an operator at, for as long as the machine had a control
2367
+ session at all. This is the *instance* question, deliberately not
2368
+ :func:`run_id_aliases_control_session`'s shape question: on tmux a
2369
+ `main`-created run `ctl-<16 hex>` owns a genuine agent session distinct
2370
+ from the fixed name (measured: killing it exactly leaves `froid-loop-ctl`
2371
+ alive), and the shape discount destroyed its run dir without ever querying
2372
+ the mux. A namespace probe that cannot answer degrades to the fixed name
2373
+ alone — the *smaller* discount, which blocks more, the safe direction. A
2374
+ discount, not a proof source: it removes non-evidence, and never clears a
2375
+ removal on transport testimony.
2376
+
2377
+ **Transport-owned name comparison.** Every comparison goes through
2378
+ :meth:`session_name_key`, never a constant fold: psmux resolves names
2379
+ through a case-folding store, tmux is case-sensitive (both measured), and
2380
+ a constant ``.lower()`` discounted a persisted `CTL` run's genuinely live
2381
+ uppercase agent on tmux as "the control session" and deleted its run dir.
2382
+ On tmux the key is identity, so the listing and tag reads keep their
2383
+ historical exact comparison. Selecting that backend is itself part of the
2384
+ listing read — :func:`mux_sessions` selects inside the caught call — so it
2385
+ degrades the listing's way: a transport that cannot even be chosen (a
2386
+ persisted `[mux] backend` naming a backend no longer registered) reports
2387
+ no live session rather than aborting every removal path."""
2388
+ try:
2389
+ mux = get_multiplexer()
2390
+ except MultiplexerError:
2391
+ return False
2392
+ key = mux.session_name_key
2393
+ name = session_name(run_id)
2394
+ control = {CTL_SESSION}
2395
+ try:
2396
+ control.add(ctl_session_for(project, mux))
2397
+ except MultiplexerError:
2398
+ pass # namespace unanswerable: only the fixed name is knowable
2399
+ if key(name) in {key(c) for c in control}:
2400
+ return False
2401
+ try:
2402
+ if key(name) not in {key(s) for s in mux_sessions()}:
2403
+ return False
2404
+ except MultiplexerError:
2405
+ return False
2406
+ try:
2407
+ tags = session_project_tags()
2408
+ except MultiplexerError:
2409
+ tags = {} # unread is not proof of foreign
2410
+ tag = next((v for s, v in tags.items() if key(s) == key(name)), "")
2411
+ return not tag or tag in accepted_tags(project)
2412
+
2413
+
2414
+ def _refuse_live_session(project: Path, run_id: str, verb: str) -> None:
2415
+ """Backstop for #419: refuse to remove a run dir out from under a live session.
2416
+
2417
+ Every caller's live guard is keyed on *engine pid* liveness, so an orphan —
2418
+ engine dead, agent session still alive in the multiplexer — passes all of them.
2419
+ That is the one state where the run dir is load-bearing: for an untagged
2420
+ session it is the only ownership proof :func:`prunable_sessions` can read, so
2421
+ removing it leaks the session (and its server) for the life of the machine.
2422
+ Refusing is a repair-path write failing loudly, per the module doctrine.
2423
+
2424
+ Scoped to what it can justify: a session this project can prove is another
2425
+ one's does not block anything (see :func:`live_session_may_be_ours`). Refusing
2426
+ there would strand nothing and wedge every removal path — including `clean`,
2427
+ which has no override — for as long as the other project's run lives.
2428
+
2429
+ Never a kill from here: a session name carries no project, so killing
2430
+ `froid-loop-<id>` by name would tear down another project's live run whenever the
2431
+ two share a run id (reachable — `--run-id` is caller-supplied).
2432
+
2433
+ The message names `froid-loop cleanup` as the remedy but does not call it sound.
2434
+ `prune_sessions` proves ownership from the tag when there is one and falls back
2435
+ to *this same run dir* when there is not — the weak proof this guard exists to
2436
+ protect, so on the untagged case it can prune another project's session on a
2437
+ shared run id (#419's second edge, pinned by
2438
+ `test_prunable_sessions_claims_an_untagged_session_on_a_run_id_collision`).
2439
+ Hence the message asks the operator to confirm first: nothing available here can
2440
+ prove the session ours, and minting a proof that outlives the run dir is #419
2441
+ direction (2), not this guard."""
2442
+ if live_session_may_be_ours(project, run_id):
2443
+ raise LiveSessionError(
2444
+ f"run {run_id}: refusing to {verb} its directory while its agent session is "
2445
+ f"still live — for an untagged session this directory is the only ownership "
2446
+ f"proof a later prune has. Clear the session with `froid-loop cleanup` first, "
2447
+ f"having confirmed it is this project's (`froid-loop attach {run_id}`): an "
2448
+ f"untagged session is proven ours by this same directory, so a run id shared "
2449
+ f"with another project would prune theirs"
2450
+ )
2451
+
2452
+
2453
+ def _discard_state_dir(project: Path, run_id: str) -> None:
2454
+ """Remove the run's out-of-tree control-plane counterpart, best-effort.
2455
+
2456
+ The events channel (#494) lives outside the project tree, so removing a run
2457
+ dir no longer removes everything the run owns: without this every
2458
+ delete/archive would leak ``<state root>/<project>/<run-id>/`` forever. It
2459
+ lives here rather than in the CLI so every caller inherits it — `delete`,
2460
+ `archive`, `clean`, the TUI's removal actions and the engine's own
2461
+ finish-time reclamation alike.
2462
+
2463
+ A **never-raise tail**, per the teardown doctrine (#139): the run dir is
2464
+ already gone by the time this runs, and failing the operator's delete over an
2465
+ unreachable state root would report a removal that in fact happened. Every
2466
+ catchable outcome is "the counterpart could not even be named" —
2467
+ :class:`StateRootError` for an environment with no derivable root, and
2468
+ ``OSError``/``RuntimeError`` for a project path the OS cannot canonicalize
2469
+ (:func:`project_tag` resolves before digesting). ``RuntimeError`` is not
2470
+ optional there: below 3.13 ``Path.resolve`` reports a symlink loop that way
2471
+ rather than as ``OSError`` (measured — 3.11 and 3.12 raise, 3.13 and 3.14
2472
+ return the unresolved path), so on two supported interpreters an ``OSError``
2473
+ -only guard lets a loop escape and breaks the promise in this paragraph.
2474
+ Removal failures are absorbed by ``ignore_errors``. Either way the orphan
2475
+ sweep in :func:`reconcile_orphan_state_dirs` is the backstop.
2476
+
2477
+ Deliberately not called by :func:`trim_run_dir`: a trimmed run is still live
2478
+ on disk and resumable, and its control plane must outlive the scaffolding.
2479
+ """
2480
+ try:
2481
+ target = state_dir_for(project, run_id)
2482
+ except (StateRootError, OSError, RuntimeError):
2483
+ return
2484
+ shutil.rmtree(target, ignore_errors=True)
2485
+
2486
+
2487
+ def _refuse_uncontained_run_dir(project: Path, run_dir: Path, action: str) -> None:
2488
+ """Refuse to remove anything but a direct child of ``project``'s runs dir.
2489
+
2490
+ The containment half of #480, and deliberately independent of how the ref was
2491
+ spelled: :func:`_is_path_escape` gates the *string* an operator typed, this
2492
+ gates the *path* the two destructive writes are about to hand `shutil.rmtree`.
2493
+ Both are wanted. `delete_run` and `archive_run` are module-public and take a
2494
+ `run_dir` outright, so a caller that composed one by some route other than
2495
+ :func:`resolve_run_dir` — the TUI's selection, a record read back from disk, a
2496
+ call site not yet written — never passes the ref guard at all.
2497
+
2498
+ ``run_dir_for`` is the sole builder of these paths, so recomposing one from
2499
+ the basename and comparing is exactly the "is a direct child" question: the
2500
+ runs root itself, a nested grandchild, and anything outside the project all
2501
+ differ from what it returns. Comparing against the rebuild rather than
2502
+ walking `parents` keeps this tracking `RUNS_DIR` the way
2503
+ :func:`_project_of_run_dir` does. The rebuild has one blind spot the name
2504
+ check closes: ``.name`` of ``runs / ".."`` is ``".."`` and the rebuild
2505
+ reproduces it verbatim, so the lexical equality holds while `rmtree` would
2506
+ resolve it to ``.froid-loop`` itself. pathlib drops ``"."`` at parse so only
2507
+ the ``".."`` spelling survives to here; ``"."`` is refused anyway rather than
2508
+ reasoned about.
2509
+
2510
+ The link walk below the equality check refuses a REDIRECTED spelling of a
2511
+ contained path: with ``.froid-loop``, ``runs`` or the run dir itself replaced
2512
+ by a symlink (or, on Windows, an unelevated ``mklink /J`` junction — why this
2513
+ is :func:`is_link_like` and not ``is_symlink``), the rebuild is lexically
2514
+ identical while `rmtree` follows the redirect and removes a tree outside the
2515
+ project. A planted redirect is this module's live threat class (see the #591
2516
+ notes in :func:`archive_run`). The walk stops short of ``project`` — the
2517
+ operator's own argument, and a project addressed through a symlinked home is
2518
+ legitimate — and covers only the orchestrator-owned levels under it. It is
2519
+ check-then-act, not fd-anchored like `journal.py`'s writes: `resolve()` is
2520
+ banned here (it can raise on a WSL-UNC host — `tests/conftest.py`'s
2521
+ ``refuse_to_resolve``), `tarfile` cannot take a dir fd at all, and the racer
2522
+ that could re-plant between check and rmtree is a live session, which the
2523
+ guard below this one refuses anyway.
2524
+
2525
+ Raises rather than degrading — observation may degrade, a repair write must
2526
+ not: there is no partial `rmtree` to fall back to, and declining quietly would
2527
+ report a removal that never happened. :class:`UnconfinedWriteError` is the
2528
+ shape-refusal this module already raises for the same class of mistake (see
2529
+ :func:`_project_of_run_dir`), and being an ``OSError`` it lands in the
2530
+ handling callers already have for a removal that failed."""
2531
+ if run_dir.name in (".", "..") or run_dir_for(project, run_dir.name) != run_dir:
2532
+ raise UnconfinedWriteError(
2533
+ f"refusing to {action} {run_dir}: not a run directory under {project / RUNS_DIR}"
2534
+ )
2535
+ node = run_dir
2536
+ while node != project:
2537
+ if is_link_like(node):
2538
+ raise UnconfinedWriteError(
2539
+ f"refusing to {action} {run_dir}: {node} is a symlink or junction"
2540
+ )
2541
+ parent = node.parent
2542
+ if parent == node: # anchored: never walk past the filesystem root
2543
+ break
2544
+ node = parent
2545
+
2546
+
2547
+ def delete_run(project: Path, run_dir: Path, *, force: bool = False) -> None:
2548
+ """Permanently remove a run directory. Callers enforce the engine-liveness
2549
+ guard; the session guard is enforced here (see :func:`_refuse_live_session`),
2550
+ which raises :class:`LiveSessionError` instead of removing.
2551
+
2552
+ ``force`` is the operator's explicit override and skips that guard, accepting
2553
+ the leak on their own say-so. It deliberately does not kill the session
2554
+ instead — that would be unscoped, and this project cannot prove the session is
2555
+ its own (which is the whole defect). Trading a possible leak of our own session
2556
+ for a possible kill of someone else's is the wrong direction for an override.
2557
+
2558
+ The containment guard runs first and is NOT under ``force``: an override is
2559
+ the operator accepting a leaked session, never a licence to rmtree a path
2560
+ outside the runs dir."""
2561
+ _refuse_uncontained_run_dir(project, run_dir, "delete")
2562
+ if not force:
2563
+ _refuse_live_session(project, run_dir.name, "delete")
2564
+ shutil.rmtree(run_dir)
2565
+ # after the run dir, never before: a raise above leaves the run whole, and a
2566
+ # whole run keeps its control plane (see _discard_state_dir).
2567
+ _discard_state_dir(project, run_dir.name)
2568
+
2569
+
2570
+ def archive_run(project: Path, run_dir: Path, *, force: bool = False) -> Path:
2571
+ """Compress a run dir into .froid-loop/archive/<id>.tar.gz and remove the
2572
+ original. The tarball is written to a temp path then atomically replaced into
2573
+ place so a partial archive never appears. Callers enforce the engine-liveness
2574
+ guard; the session guard is enforced here (see :func:`_refuse_live_session`,
2575
+ and :func:`delete_run` for ``force``) and runs before the tarball is written,
2576
+ so a refusal leaves nothing behind.
2577
+
2578
+ The tarball holds the run dir only, so since #494 an archive no longer carries
2579
+ the run's ``events/``: the channel moved out of the tree, and its files are
2580
+ transient completion signals the watcher has already consumed — the recorded
2581
+ decision accepts losing them from the archive. Everything an archive is read
2582
+ for later (state, journal, tasks, logs) is in the run dir and unaffected.
2583
+
2584
+ Containment (see :func:`_refuse_uncontained_run_dir`) is checked ahead of both,
2585
+ for the reason the session guard runs early: a refusal must leave no archive
2586
+ directory and no tarball behind."""
2587
+ _refuse_uncontained_run_dir(project, run_dir, "archive")
2588
+ if not force:
2589
+ _refuse_live_session(project, run_dir.name, "archive")
2590
+ archive_dir = project / ARCHIVE_DIR
2591
+ archive_dir.mkdir(parents=True, exist_ok=True)
2592
+ dest = archive_dir / f"{run_dir.name}.tar.gz"
2593
+ # #363: the guard, not a helper — the path is handed to `tarfile.open`, so there
2594
+ # is no payload for `atomic_write_*` to take. Nothing gitignores this directory:
2595
+ # init writes `.froid-loop/runs/`, `.froid-loop/cache/`, `.froid-loop/policy.toml`
2596
+ # and `_froid/render/`, and `archive/` matches none of them. So a stranded temp
2597
+ # here is an untracked file holding `worktree_clean` False until a human removes
2598
+ # it — the same exposure `decisions._write_store`, `policy.write_mux_backend` and
2599
+ # `tui.settings.PolicyDoc.save` had. (Not the sweep's two `decisions.json`
2600
+ # writes, which look like the same fix but write under the ignored run dir.)
2601
+ #
2602
+ # #591: staged through `_mkstemp_beside` — the atomic writers' own exclusive
2603
+ # `0600` create (binary-mode on win32), under a fresh unpredictable name per
2604
+ # attempt. A fixed name made a temp stranded by a kill, or planted at the
2605
+ # guessable spelling, deny every later attempt as `FileExistsError`; the
2606
+ # truncate-and-reuse it replaced followed a planted symlink instead. mkstemp's
2607
+ # exclusivity still never opens a name something else holds, and the name being
2608
+ # this process's own mint is what licenses the cleanup unlink below. It sits
2609
+ # outside the `try` on purpose: a create that fails has staged nothing to
2610
+ # clean up.
2611
+ fd, tmp_name = _mkstemp_beside(dest)
2612
+ tmp = Path(tmp_name)
2613
+ try:
2614
+ with os.fdopen(fd, "wb") as raw:
2615
+ with tarfile.open(fileobj=raw, mode="w:gz") as tar:
2616
+ tar.add(run_dir, arcname=run_dir.name)
2617
+ # Flushed and fsynced before the publish, and unlike the rest of this
2618
+ # family that is not about staleness but about data loss: `shutil.rmtree`
2619
+ # below removes the only other copy of the run, so a crash with the
2620
+ # tarball still in page cache destroys it outright. Ordered inside the
2621
+ # fdopen context so the gzip trailer `tar.close()` just wrote is included.
2622
+ raw.flush()
2623
+ os.fsync(raw.fileno())
2624
+ atomic_replace(tmp, dest)
2625
+ except BaseException:
2626
+ with contextlib.suppress(OSError):
2627
+ tmp.unlink(missing_ok=True) # provably ours: mkstemp minted the name
2628
+ raise
2629
+ shutil.rmtree(run_dir)
2630
+ _discard_state_dir(project, run_dir.name) # same tail as delete_run
2631
+ return dest
2632
+
2633
+
2634
+ # ------------------------------------------------------- reclaim / retention
2635
+
2636
+ # Heavy per-run scaffolding trimmed from a concluded run dir while the
2637
+ # TUI-visible core (state.json, journal.jsonl, logs/, ATTENTION) is preserved,
2638
+ # so the run still lists and renders in the dashboard. "worktrees" mirrors
2639
+ # workspace.WORKTREE_DIRNAME; kept literal here to avoid an import cycle
2640
+ # (workspace imports nothing from runs, but runs stays leaf-light on purpose).
2641
+ #
2642
+ # VERIFY_DIR is the retained verifier stdout/stderr store. It qualifies as heavy
2643
+ # on the same measure as a worktree checkout: `[verify] stream_capture_kb`
2644
+ # defaults to 256 KiB per stream, so a run accumulates up to 512 KiB per verify
2645
+ # command per attempt, and nothing else ever reclaims it. Its journal records
2646
+ # survive the trim and keep naming the files (`stdout_path`/`stderr_path`), which
2647
+ # is the same bargain `worktrees` already makes — a trimmed run is a run you can
2648
+ # still see and resume, not one you can still re-read every artifact of. Imported
2649
+ # from the writer rather than re-spelled, so the reclaim cannot drift from the
2650
+ # directory `Journal.write_verify_stream` actually creates.
2651
+ _HEAVY_RUN_ENTRIES = ("worktrees", VERIFY_DIR)
2652
+
2653
+
2654
+ def heavy_run_entries(run_dir: Path) -> list[Path]:
2655
+ """The paths :func:`trim_run_dir` would remove from ``run_dir``.
2656
+
2657
+ Exists so a caller sizing the reclaim measures exactly what the trim takes.
2658
+ `clean` sums these before mutating (its estimate has to hold under
2659
+ --dry-run); reading the tuple through this function is what keeps that sum
2660
+ from silently going stale the next time an entry is added to it."""
2661
+ return [run_dir / name for name in _HEAVY_RUN_ENTRIES]
2662
+
2663
+
2664
+ def _state_or_none(run_dir: Path):
2665
+ """Parsed run state, or None when it cannot be read — never classify (and so
2666
+ never reclaim) what you cannot positively read."""
2667
+ try:
2668
+ return load_state(run_dir)
2669
+ except Exception: # unreadable/corrupt state ⇒ leave it alone
2670
+ return None
2671
+
2672
+
2673
+ def is_finished(run_dir: Path) -> bool:
2674
+ """A finished, no-longer-live run. `resume` refuses these (cli checks
2675
+ state.finished), so tearing down their worktrees can never strand a resume —
2676
+ the safe predicate for the *automatic* reconcile paths."""
2677
+ if engine_alive(run_dir):
2678
+ return False
2679
+ state = _state_or_none(run_dir)
2680
+ return bool(state and state.finished)
2681
+
2682
+
2683
+ def reclaimable(run_dir: Path) -> bool:
2684
+ """A terminal run (finished or stopped) with no live engine — eligible for
2685
+ the *explicit* `clean` command. A stopped run is technically resumable, so
2686
+ reclaiming its worktree ends that; `clean` is an opt-in reclaim (guarded by
2687
+ --keep / --dry-run). Paused, interrupted (crashed) and running/unknown-host
2688
+ runs are never reclaimed: paused/interrupted are actively resumable, and a
2689
+ missing pid could mean a foreign-host run, so we require positive local
2690
+ termination evidence (finished or stopped)."""
2691
+ if engine_alive(run_dir):
2692
+ return False
2693
+ state = _state_or_none(run_dir)
2694
+ return bool(state and (state.finished or state.stopped))
2695
+
2696
+
2697
+ def reconcile_orphan_worktrees(repo: Path, run_dir: Path, *, dry_run: bool = False) -> list[Path]:
2698
+ """Force-remove every git worktree whose path lies under ``run_dir``, then
2699
+ prune git's admin entries. Reconciles from ``git worktree list`` (on-disk
2700
+ truth), NOT from policy — orphans created under a previous isolation=worktree
2701
+ config persist after a switch back to isolation=none. Returns the worktree
2702
+ paths handled (or that would be, under dry_run). Callers gate on
2703
+ ``reclaimable``; the main checkout is never under a run dir, so it is safe."""
2704
+ run_res = run_dir.resolve()
2705
+ try:
2706
+ worktrees = verify.worktree_list(repo)
2707
+ except verify.GitError:
2708
+ return []
2709
+ handled: list[Path] = []
2710
+ for wt in worktrees:
2711
+ try:
2712
+ wt.resolve().relative_to(run_res)
2713
+ except (ValueError, OSError):
2714
+ continue # not this run's worktree (incl. the main checkout)
2715
+ handled.append(wt)
2716
+ if not dry_run:
2717
+ try:
2718
+ verify.worktree_remove(repo, wt, force=True)
2719
+ except verify.GitError:
2720
+ shutil.rmtree(wt, ignore_errors=True)
2721
+ if handled and not dry_run:
2722
+ verify.worktree_prune(repo)
2723
+ return handled
2724
+
2725
+
2726
+ def reconcile_stale_worktrees(repo: Path, project: Path, *, dry_run: bool = False) -> list[Path]:
2727
+ """Safety net for the automatic paths (run/sweep start): tear down worktrees
2728
+ left behind by a *finished* run whose clean-finish GC didn't complete (e.g. a
2729
+ crash between merge and teardown). Deliberately finished-ONLY — a stopped run
2730
+ is still resumable, so its worktree is left for `resume`/`clean` to handle and
2731
+ never stranded out from under the operator."""
2732
+ handled: list[Path] = []
2733
+ for run_dir in list_run_dirs(project):
2734
+ if not is_finished(run_dir):
2735
+ continue
2736
+ handled += reconcile_orphan_worktrees(repo, run_dir, dry_run=dry_run)
2737
+ return handled
2738
+
2739
+
2740
+ def _run_dir_names(project: Path) -> set[str] | None:
2741
+ """Every *directory name* under the runs dir, or ``None`` when that listing
2742
+ could not be taken.
2743
+
2744
+ Deliberately not :func:`list_run_dirs`, which is ``state.json``-gated: this
2745
+ answers "does a run dir by this name exist", and a run whose ``state.json`` is
2746
+ missing or corrupt still owns its control plane. Gating on state.json would
2747
+ sweep the counterpart out from under exactly the run an operator is trying to
2748
+ recover.
2749
+
2750
+ The two failures are distinguished because they mean opposite things. A
2751
+ *missing* runs dir is a real answer — no runs, so nothing is live — while an
2752
+ unreadable one answers nothing at all, and a sweep run against "no live names"
2753
+ would remove every state dir this project has. ``None`` is that second case.
2754
+ """
2755
+ try:
2756
+ return {entry.name for entry in os.scandir(project / RUNS_DIR) if entry.is_dir()}
2757
+ except FileNotFoundError:
2758
+ return set()
2759
+ except OSError:
2760
+ return None
2761
+
2762
+
2763
+ def reconcile_orphan_state_dirs(project: Path, *, dry_run: bool = False) -> list[Path]:
2764
+ """Remove this project's out-of-tree control-plane dirs whose run dir is gone.
2765
+
2766
+ The GC backstop for the events channel (#494). :func:`_discard_state_dir`
2767
+ removes the counterpart on every ordinary delete/archive, so this catches what
2768
+ that path could not: a run dir removed by hand or by an `rm -rf .froid-loop`,
2769
+ a delete that ran before this version existed, and any tail that failed
2770
+ quietly. Without it the state root accumulates one dead subtree per run
2771
+ forever, on a path outside the project that no operator thinks to look at.
2772
+
2773
+ Shaped like :func:`reconcile_orphan_worktrees`: enumerate on-disk truth,
2774
+ containment-test each path, remove with failures tolerated. Returns what was
2775
+ removed (or, under ``dry_run``, what would be).
2776
+
2777
+ Every path is built from an entry name this function itself enumerated —
2778
+ never from a caller-supplied ref, which is what :func:`_is_path_escape`
2779
+ refuses on the ref-resolution path. Entries that are not real directories are
2780
+ skipped, symlinks included: a link is not a state dir we created, and
2781
+ reporting one swept would be a false count even where ``rmtree`` refuses it.
2782
+ The containment test then covers what ``is_symlink`` cannot — a Windows
2783
+ *junction* reads as a plain directory while ``resolve()`` follows it, so
2784
+ without the test ``rmtree`` would empty a target sitting outside the root.
2785
+ That case is POSIX-invisible, and the tests say so rather than claim it.
2786
+
2787
+ Degrades to no-op rather than raising, in either direction: an underivable
2788
+ state root, an unreadable root, or an unreadable runs dir all sweep nothing.
2789
+ This is reclamation, not repair — leaving disk behind is the cheap outcome,
2790
+ and removing a live run's control plane is not.
2791
+
2792
+ Both guards hold ``RuntimeError`` alongside ``OSError`` for the same reason
2793
+ :func:`_discard_state_dir` does: every path here is resolved (the project by
2794
+ :func:`project_tag`, then the root, then each entry), and below 3.13
2795
+ ``Path.resolve`` reports a symlink loop as ``RuntimeError``. A loop planted
2796
+ among the entries would otherwise escape a sweep whose whole contract is to
2797
+ degrade, and take the operator's ``clean`` down with it after its real work
2798
+ was already done.
2799
+
2800
+ **The two reads are ordered, and the order is the whole race guard.** State
2801
+ entries are enumerated *before* the live run-dir names, because a run creates
2802
+ its run dir strictly before its state dir — ``compose_run`` builds the
2803
+ ``Journal`` (which mkdirs the run dir) and only then stamps the config digest
2804
+ (:func:`write_trusted_config_digest`, the earliest writer into the state dir
2805
+ since #498) and calls ``make_adapters``, whose ``SignalWatcher`` mkdirs the
2806
+ events dir alongside it. Reading entries first makes
2807
+ that ordering carry the guarantee: anything in ``entries`` had its state dir
2808
+ on disk at the first read, so its run dir was on disk *before* that, so the
2809
+ later ``live`` read is certain to contain it. Read the other way round, a run
2810
+ starting in the gap is missing from ``live`` and present in ``entries``, and
2811
+ an operator's ``clean`` deletes the control plane of a run that is starting
2812
+ right now — whose watcher then polls a primary that no longer exists, or
2813
+ simply never sees the Stop. A run dir that disappears *between* the reads is
2814
+ the opposite case and correctly swept: it is a real orphan by then.
2815
+ """
2816
+ try:
2817
+ root = project_state_root(project)
2818
+ entries = sorted(root.iterdir())
2819
+ root_res = root.resolve()
2820
+ except (StateRootError, OSError, RuntimeError):
2821
+ return []
2822
+ live = _run_dir_names(project)
2823
+ if live is None:
2824
+ return []
2825
+ handled: list[Path] = []
2826
+ for entry in entries:
2827
+ if entry.name in live or entry.is_symlink() or not entry.is_dir():
2828
+ continue
2829
+ if entry.name == MUX_REGISTRY_DIR:
2830
+ # Not a run entry at all (`mux_registry_root`), and the one entry here
2831
+ # whose deletion costs more than the disk it reclaims: it holds the
2832
+ # `.port`/`.key` files every psmux verb resolves a session through, so
2833
+ # sweeping it while a server is up leaves that server alive,
2834
+ # unreachable, and invisible to `psmux ls` in any registry — the
2835
+ # manufactured orphan the root was moved out of the project tree to
2836
+ # avoid. Never reaped rather than reaped-when-empty: proving it empty
2837
+ # means asking every server in it whether it is alive, and this sweep
2838
+ # has no seam to the multiplexer (nor may it acquire one — it must
2839
+ # degrade to a no-op, and a transport probe cannot promise that).
2840
+ # psmux removes its own quartet on session shutdown, so what is left
2841
+ # behind is a directory of small files, not growth.
2842
+ continue
2843
+ try:
2844
+ entry.resolve().relative_to(root_res)
2845
+ except (OSError, RuntimeError, ValueError):
2846
+ continue
2847
+ handled.append(entry)
2848
+ if not dry_run:
2849
+ shutil.rmtree(entry, ignore_errors=True)
2850
+ return handled
2851
+
2852
+
2853
+ def _unlink_redirect(p: Path) -> None:
2854
+ """Remove a link-like entry itself, never what it points at.
2855
+
2856
+ ``shutil.rmtree`` REFUSES a directory symlink by design (it would otherwise
2857
+ delete the target's contents), and under ``ignore_errors=True`` that refusal
2858
+ is swallowed — so trimming a planted redirect reported success while leaving
2859
+ the link on disk. Unlink covers a POSIX symlink and a win32 file symlink;
2860
+ ``rmdir`` is the win32 arm, where ``DeleteFileW`` rejects a directory symlink
2861
+ or junction and ``RemoveDirectoryW`` drops the reparse point without
2862
+ following it. Best-effort to match the ``rmtree`` beside it: a trim is
2863
+ reclamation, and a run dir we cannot fully reclaim is not a reason to abort
2864
+ the whole `clean`."""
2865
+ try:
2866
+ p.unlink()
2867
+ except OSError:
2868
+ with contextlib.suppress(OSError):
2869
+ p.rmdir()
2870
+
2871
+
2872
+ def trim_run_dir(run_dir: Path, *, dry_run: bool = False) -> list[Path]:
2873
+ """Delete heavy scaffolding (the ``worktrees/`` tree and the retained
2874
+ verifier stream store) from a concluded run dir, preserving its TUI-visible
2875
+ core so the run still appears in the dashboard with full status/journal/logs.
2876
+ Returns the paths removed.
2877
+
2878
+ The run's out-of-tree control plane is deliberately left alone (see
2879
+ :func:`_discard_state_dir`): a trimmed run still exists and is still
2880
+ resumable, so its state dir has to outlive its scaffolding."""
2881
+ removed: list[Path] = []
2882
+ for p in heavy_run_entries(run_dir):
2883
+ link = is_link_like(p)
2884
+ if not (p.exists() or link):
2885
+ continue
2886
+ removed.append(p)
2887
+ if dry_run:
2888
+ continue
2889
+ if link:
2890
+ _unlink_redirect(p)
2891
+ else:
2892
+ shutil.rmtree(p, ignore_errors=True)
2893
+ return removed
2894
+
2895
+
2896
+ def _run_started_epoch(run_dir: Path) -> float | None:
2897
+ """Unix time parsed from the run id's ``YYYYMMDD-HHMMSS`` prefix, or None
2898
+ when the name does not carry one (legacy/foreign id)."""
2899
+ try:
2900
+ return time.mktime(time.strptime(run_dir.name[:15], "%Y%m%d-%H%M%S"))
2901
+ except (ValueError, OverflowError):
2902
+ return None
2903
+
2904
+
2905
+ def runs_past_retention(
2906
+ run_dirs: list[Path], *, keep_n: int, keep_days: int = 0, now: float | None = None
2907
+ ) -> list[Path]:
2908
+ """The subset of ``run_dirs`` (oldest-first) beyond the retention window:
2909
+ not among the newest ``keep_n``, and — when ``keep_days`` is set — also older
2910
+ than ``keep_days`` days. ``keep_n <= 0`` retains nothing by count; an
2911
+ unparseable run id is treated as old enough to prune once past ``keep_n``."""
2912
+ ordered = list(run_dirs)
2913
+ candidates = (
2914
+ ordered[:-keep_n]
2915
+ if keep_n > 0 and len(ordered) > keep_n
2916
+ else ([] if keep_n > 0 else list(ordered))
2917
+ )
2918
+ if keep_days and keep_days > 0:
2919
+ cutoff = (time.time() if now is None else now) - keep_days * 86400
2920
+ return [rd for rd in candidates if (_run_started_epoch(rd) or 0.0) < cutoff]
2921
+ return candidates
2922
+
2923
+
2924
+ # ----------------------------------------------------------- escalation resolution
2925
+
2926
+
2927
+ class RearmError(Exception):
2928
+ """The run/story is not in a re-armable escalation state."""
2929
+
2930
+
2931
+ def validate_restore_latch(
2932
+ state: RunState, task: StoryTask, story_key: str, *, worktree_isolation: bool = False
2933
+ ) -> str | None:
2934
+ """Every precondition an intent-gap patch-restore latch (Froid Plane #2564) must
2935
+ satisfy, in one place. Returns an operator-facing error string, or None to latch.
2936
+
2937
+ The single seam for both entry points: `rearm_escalation` (which performs the
2938
+ latch, and is also reachable programmatically — a TUI restore, a future caller)
2939
+ and `cli._resolve_restore_patch` (which fails fast *before* the interactive
2940
+ resolve session, so an unhonorable restore doesn't cost an agent conversation).
2941
+ Splitting these let a non-CLI caller bypass the worktree half; keeping them here
2942
+ means a caller cannot latch a patch the engine could never honor.
2943
+
2944
+ The CLI knows one thing this cannot: the *live* policy's isolation mode, which
2945
+ may have been edited between escalation and resolve. It passes that as
2946
+ `worktree_isolation`; the recorded `task.worktree_path` (how the unit actually
2947
+ executed) is checked here either way, so both entry points reject a
2948
+ worktree-isolation restore and the CLI additionally catches a policy flip.
2949
+
2950
+ Path resolution and trusted-roots containment stay CLI-side: they need
2951
+ `--project` and the loaded froid config, neither of which run state carries.
2952
+ """
2953
+ # A sentinel-wedged story escalated BEFORE planning — there is no attempted
2954
+ # implementation to restore, and its re-arm re-dispatches a planning leg.
2955
+ # Keyed on the recorded detection verdict (task.sentinel_kind), not the on-disk
2956
+ # basename, mirroring rearm_escalation's sentinel-clear branch.
2957
+ if state.source == "stories" and task.sentinel_kind:
2958
+ return (
2959
+ f"story {story_key} is wedged on a pre-planning {task.sentinel_kind} sentinel — "
2960
+ "there is no attempted implementation to restore, and the re-drive starts "
2961
+ "at planning. Re-run resolve without a restore patch for a clean re-plan."
2962
+ )
2963
+ # Same seam, broader shape: a restore only works through the spec's in-review
2964
+ # flip, so an escalation with NO recorded spec (an ambiguous two-file wedge, an
2965
+ # unknown --story selector, a session that died before naming one) has no
2966
+ # routing target — the latch would stick, the flip would be skipped, and the
2967
+ # engine would lay the patch onto the tree before a planning leg.
2968
+ if not task.spec_file:
2969
+ return (
2970
+ f"story {story_key} has no recorded spec file, so a restored patch has no "
2971
+ "review to resume (the re-drive starts at planning). Re-run resolve "
2972
+ "without a restore patch for a from-scratch re-drive."
2973
+ )
2974
+ # Restore is an in-place-only recovery: a worktree-isolation re-drive discards
2975
+ # the unit's worktree (engine._finish_inflight — taking a patch saved inside it
2976
+ # along) and re-mounts a fresh one, so the re-apply could only fail on a
2977
+ # destroyed patch file. Reject up front instead of latching a patch that can
2978
+ # never restore.
2979
+ if worktree_isolation or task.worktree_path:
2980
+ return (
2981
+ "restore patch is unsupported for worktree-isolation runs (the re-drive "
2982
+ "discards and re-mounts the unit's worktree, so an in-place restore has "
2983
+ "nothing durable to land on) — re-arm from scratch instead: drop "
2984
+ "--restore-patch, or if the resolve agent recorded the restore in "
2985
+ "resolution.json, re-run with --no-interactive (which ignores that "
2986
+ "marker) instead of repeating the agent session"
2987
+ )
2988
+ return None
2989
+
2990
+
2991
+ def task_spec_path(task: StoryTask, state: RunState) -> Path:
2992
+ """The recorded spec path, re-anchored on the tree it was persisted relative to.
2993
+
2994
+ `StoryTask._serialized_worktree_path` (`model.py`) persists a worktree-local spec
2995
+ RELATIVE to the mounted worktree root, and `from_dict` reads it back raw. Resolving
2996
+ that against the process cwd is not merely unreachable — it is actively wrong:
2997
+ `froid-loop resolve` runs from the project root, where the MAIN CHECKOUT carries the
2998
+ same `_froid-output/specs/...` layout, so a bare `Path(task.spec_file)` names the main
2999
+ checkout's copy of the story spec. `is_file()` then answers True, `confine_root`
3000
+ accepts it (it genuinely is under `project`), and the status flip and the baseline
3001
+ re-stamp both land on a file the run never used while the worktree's real spec is
3002
+ left on the escalated attempt's sha.
3003
+
3004
+ Absolute paths pass through: a spec outside the worktree is persisted verbatim.
3005
+
3006
+ Raises `ValueError` on an empty `task.spec_file` rather than documenting a
3007
+ precondition nothing enforces: `Path("")` is `.`, so `root / raw` would answer the
3008
+ ROOT DIRECTORY — a write target, not a spec. Every caller already guards; this is
3009
+ public now, so the next one gets an exception instead of a silent tree root.
3010
+ """
3011
+ if not task.spec_file:
3012
+ raise ValueError("task_spec_path requires a non-empty task.spec_file")
3013
+ raw = Path(task.spec_file)
3014
+ if raw.is_absolute():
3015
+ return raw
3016
+ return task_spec_root(task, state) / raw
3017
+
3018
+
3019
+ def task_spec_root(task: StoryTask, state: RunState) -> Path:
3020
+ """The tree a `task.spec_file` is anchored on — and confined to.
3021
+
3022
+ One definition backs both halves because they must not disagree: the root
3023
+ `task_spec_path` resolves against and the `confine_root` the writers validate the
3024
+ result against are the same claim about which tree owns this spec. Passing
3025
+ `state.project` while resolving against the worktree does not REFUSE the mismatch —
3026
+ `set_frontmatter_status`, `devcontract.strip_auto_run_result` and
3027
+ `verify.set_frontmatter_field` all answer an out-of-root path by silently dropping
3028
+ to the plain no-follow write, losing the confined arm's O_NOFOLLOW walk of the
3029
+ parent components (#593) with no signal at all.
3030
+
3031
+ Worktrees normally resolve under `<project>/.froid-loop/runs/...`, so the confined
3032
+ arm is taken by construction rather than by luck — no policy or env var can
3033
+ relocate them. The one escape is that `workspace.open_unit_workspace` stores a
3034
+ `.resolve()`d path: a symlinked `.froid-loop`, `runs` or `worktrees` lands the spec
3035
+ outside `project`, and before this anchor moved that silently degraded all three
3036
+ writes.
3037
+
3038
+ A worktree that CANNOT confine the anchored path yields the project instead. An
3039
+ absolute `spec_file` beside a set `worktree_path` is precisely the out-of-mount
3040
+ shape: `model._serialized_worktree_path` keeps a path verbatim exactly when
3041
+ `relative_to(worktree_path)` raises, so the two spellings did not share a prefix.
3042
+ Returning the worktree there would name a root that can never contain the path
3043
+ `task_spec_path` passes through — the three `_atomic_write_spec` writers would
3044
+ silently take the plain no-follow arm (losing #593's O_NOFOLLOW walk) and
3045
+ `_restore_rearmed_spec`, which calls the confined writer directly, would RAISE.
3046
+
3047
+ The project can often confine it. Where nothing can, the THREE `_atomic_write_spec`
3048
+ writers land on the arm they already took — they select lexically, so an out-of-root
3049
+ path simply takes the plain no-follow write as before. That is not true of every
3050
+ writer: `_restore_rearmed_spec` calls `atomic_write_bytes_confined` DIRECTLY with no
3051
+ lexical arm, so for a spec outside both the mount and the project — the shared
3052
+ artifact dir `_spec_is_shared_with_the_redrive` treats as first-class and reachable —
3053
+ it raises `UnconfinedWriteError` and the re-arm's undo is lost with the spec already
3054
+ flipped and stripped. That asymmetry PRE-DATES this anchor (the previous body
3055
+ returned the worktree there, which equally cannot confine the path) and is tracked
3056
+ separately; it is named here so the paragraph is not read as covering it.
3057
+
3058
+ The arm is not unconditionally an improvement either, and that exception is graded by
3059
+ `test_task_spec_root_refuses_a_spec_the_project_cannot_reach`: `_atomic_write_spec`
3060
+ picks its arm on a LEXICAL `is_relative_to`, but the confined arm it picks then
3061
+ walks the components below the root and refuses a redirect (`open_dir_confined` on
3062
+ POSIX, `path_is_confined` on win32). A spec that is lexically under the project but
3063
+ reached THROUGH a symlinked component — a symlinked `_froid-output`, say — therefore
3064
+ moves from a succeeding plain no-follow write to `UnconfinedWriteError`, which
3065
+ `rearm_escalation` re-raises as `RearmError`. That is a re-arm which used to
3066
+ complete and now aborts, so this arm is not the pure improvement an earlier draft of
3067
+ this docstring claimed.
3068
+
3069
+ It is kept anyway, because the alternative is worse. Predicting the walk here (gate
3070
+ the arm on `path_is_confined` and fall back to the worktree) makes the ROOT depend
3071
+ on filesystem state: `path_is_confined` answers False for a component it cannot
3072
+ probe, so a spec whose parent does not exist yet would anchor on the worktree and
3073
+ the same spec would anchor on the project once the directory appeared. A confine
3074
+ root that moves under a `mkdir` is not a definition. The refusal is also the correct
3075
+ posture on its own terms — #593 exists to refuse writes through a link on a path
3076
+ that came from a session-driven scan — so this trades a narrow, LOUD failure for a
3077
+ deterministic rule, and the failure names the path in its message.
3078
+
3079
+ The test is the same lexical `is_relative_to` the writer gates on, so the root and
3080
+ the writer's ARM SELECTION agree by construction; only the walk below can still
3081
+ refuse. Deliberately not canonicalized: `_spec_is_shared_with_the_redrive` answers a
3082
+ DIFFERENT question (is this spec reachable by the re-drive) and canonicalizes for
3083
+ it, but matching that here would diverge from the gate this value is measured
3084
+ against and change writes that are correct today.
3085
+ """
3086
+ worktree = task.worktree_path
3087
+ if not worktree:
3088
+ return Path(state.project)
3089
+ raw = Path(task.spec_file or "")
3090
+ if raw.is_absolute() and not raw.is_relative_to(worktree):
3091
+ return Path(state.project)
3092
+ return Path(worktree)
3093
+
3094
+
3095
+ def task_stories_root(task: StoryTask | None, state: RunState) -> Path:
3096
+ """The tree this run's STORIES FOLDER lives in — the workspace root, not a
3097
+ confinement root.
3098
+
3099
+ Deliberately NOT `task_spec_root`, which the sentinel and stories-block readers
3100
+ used to borrow. That function answers "which tree can CONFINE a write to
3101
+ `task.spec_file`", and its out-of-mount arm falls back to the project precisely so
3102
+ a `confine_root` can never fail to contain the anchored path. Reusing that answer
3103
+ here imported a write-confinement decision into a READ of a different file: for an
3104
+ isolated run whose `spec_file` is absolute and lexically outside the mount — the
3105
+ shape `model._serialized_worktree_path` persists verbatim, reachable whenever a
3106
+ symlinked component makes a spec that physically lives in the mount look outside
3107
+ it, since `verify.resolve_spec_path` deliberately does not `.resolve()` — the
3108
+ stories folder would be looked up in the MAIN CHECKOUT while
3109
+ `stories_engine._stories_folder` answers the worktree for the same task. One
3110
+ surface would then describe two trees, which is the exact defect the spec anchor
3111
+ exists to close.
3112
+
3113
+ So this mirrors `_stories_folder`'s own rule instead: the mount whenever the task
3114
+ holds one, the project otherwise. `spec_file` does not enter into it — the stories
3115
+ folder is located by `state.spec_folder` relative to the workspace root, and a
3116
+ task's spec being elsewhere says nothing about where its story manifest lives.
3117
+
3118
+ A mount that is GONE degrades to the project. `worktree_path` is cleared at
3119
+ exactly one site in the engine — the restart discard — so a task that reached a
3120
+ terminal phase through successful integration keeps naming the unit worktree its
3121
+ teardown already removed. The `done_checkpoint` pause is raised in precisely that
3122
+ window, and the TUI reads this for the checkpoint card's title and description, so
3123
+ trusting the stale field lost the committed story's manifest to a deleted
3124
+ directory while the merged copy sat in the project checkout.
3125
+
3126
+ Answering on filesystem state is right HERE and would be wrong in
3127
+ `task_spec_root`: that one is a write-confinement root, where a value that moves
3128
+ under a `mkdir` is not a definition. This is a READ locator, and observation
3129
+ degrades rather than raising — a probe that cannot answer falls back to the tree
3130
+ that always exists.
3131
+
3132
+ Accepts `None` so the two call sites do not each re-spell the no-task fallback.
3133
+ """
3134
+ if task is None or not task.worktree_path:
3135
+ return Path(state.project)
3136
+ mount = Path(task.worktree_path)
3137
+ try:
3138
+ if not mount.is_dir():
3139
+ return Path(state.project)
3140
+ except OSError:
3141
+ return Path(state.project)
3142
+ return mount
3143
+
3144
+
3145
+ def _spec_is_shared_with_the_redrive(state: RunState, task: StoryTask) -> bool:
3146
+ """True when the recorded spec lives outside BOTH checkouts, so the re-arm's status
3147
+ flip survives a mount's disposal and the ISOLATED re-drive reads it.
3148
+
3149
+ Asked only of a re-drive that will mount (`spec_reaches_the_redrive`'s isolated
3150
+ arm), and deliberately not of a task that HAS a mount: those are two different
3151
+ questions, and a policy flip separates them. A run switched from `isolation = "none"`
3152
+ to `"worktree"` while an escalation is paused re-drives isolated with no mount
3153
+ recorded at all, and the recorded spec is then measured against the project alone —
3154
+ which is the whole point, since the fresh worktree is cut from git and reads no
3155
+ working tree.
3156
+
3157
+ The case: artifact dirs configured outside the project tree. `ProjectPaths.rebased`
3158
+ leaves those exactly where they are ("configured outside the project tree; doesn't
3159
+ move") — they are SHARED across checkouts, not per-worktree — so the spec the dev
3160
+ session reported resolves to one file that every worktree sees. The re-drive reads it
3161
+ back through `verify.resolve_spec_path`, whose absolute branch passes the value
3162
+ through untouched, and `engine._dispatched_spec_for_attempt` then accepts it because
3163
+ the rebased `implementation_artifacts` is still that same external directory.
3164
+
3165
+ Both roots are load-bearing, and neither implies the other:
3166
+
3167
+ - INSIDE the worktree — the file the fresh mount destroys. Unreachable.
3168
+ - inside the PROJECT but outside the worktree — the main checkout's copy. The write
3169
+ lands, but the re-drive cannot use it: under isolation `workspace.paths` is rebased
3170
+ onto the fresh worktree, so `verify.spec_within_roots` measures the main
3171
+ checkout's path against worktree-local roots and rejects it. Unreachable, and this
3172
+ is the one shape the worktree test alone would wrongly exempt.
3173
+ - outside both — the shared artifact dir above. Reachable.
3174
+
3175
+ (The two are not nested: worktrees normally sit under `<project>/.froid-loop/runs/`,
3176
+ but `workspace.open_unit_workspace` stores a `.resolve()`d path, so a symlinked
3177
+ `.froid-loop` puts the mount outside the project.)
3178
+
3179
+ The recorded spelling opens the question but does not answer it.
3180
+ `StoryTask._serialized_worktree_path` persists a spec RELATIVE whenever it sits under
3181
+ the mounted worktree (and, with no mount, whenever the run recorded it relative to
3182
+ the project), and verbatim (absolute) otherwise — so an absolute value is the only
3183
+ shape that can be shared. But that relativize is a LEXICAL `relative_to` against the
3184
+ same `worktree_path` read here, so all an absolute value proves is that the two
3185
+ spellings did not share a prefix. A spec reported through a symlink or a `..` segment
3186
+ sits inside the worktree and is persisted absolute all the same, and answering
3187
+ "shared" for it would suppress the warning on a spec that really is destroyed with
3188
+ the worktree.
3189
+
3190
+ So containment is decided on the CANONICAL paths, and a host that cannot canonicalize
3191
+ one of them answers "not shared". That degrade is the safe direction and the reason
3192
+ this does not use `resolve_or_lexical`: its fallback is `absolute()`, which does not
3193
+ fold `..`, so a spec spelled through either checkout would come back looking external
3194
+ and go silent — trading a wrong warning for no warning at all."""
3195
+ raw = Path(task.spec_file or "")
3196
+ if not raw.is_absolute():
3197
+ return False
3198
+ try:
3199
+ # the house pair — `resolve()` raises RuntimeError, not OSError, for a symlink
3200
+ # loop on the 3.11/3.12 floor
3201
+ real = raw.resolve()
3202
+ if real.is_relative_to(Path(state.project).resolve()):
3203
+ return False
3204
+ if task.worktree_path and real.is_relative_to(Path(task.worktree_path).resolve()):
3205
+ return False
3206
+ return True
3207
+ except (OSError, RuntimeError):
3208
+ return False
3209
+
3210
+
3211
+ def _spec_is_inside_the_mount(task: StoryTask) -> bool:
3212
+ """True when the file `task_spec_path` names sits INSIDE the mount this task
3213
+ recorded — so a write to it cannot reach an IN-PLACE re-drive, which reads the main
3214
+ checkout.
3215
+
3216
+ The mirror of `_spec_is_shared_with_the_redrive`, for the other arm of
3217
+ `spec_reaches_the_redrive`. Reachable only through a policy flip: a run switched
3218
+ from `isolation = "worktree"` to `"none"` while an escalation is paused still
3219
+ carries the escalated attempt's `worktree_path`, so `task_spec_path` re-anchors the
3220
+ edit on that mount while `engine._run_story` re-runs the story in the main checkout.
3221
+ `_finish_inflight` releases the mount-owned spelling at RESUME, which is after
3222
+ `froid-loop resolve` has already written the context and re-armed — this is what the
3223
+ human and the agent are told in the meantime.
3224
+
3225
+ Unlike the shared test, containment inside the PROJECT is not disqualifying: an
3226
+ in-place re-drive reads the main checkout's working tree, so a spec anywhere the
3227
+ project can see it reaches. Only the mount is out of reach.
3228
+
3229
+ A relative spelling beside a recorded mount is inside it BY CONSTRUCTION —
3230
+ `_serialized_worktree_path` relativizes exactly when `relative_to(worktree_path)`
3231
+ succeeds — so it needs no filesystem probe and gets none. Absolute spellings are
3232
+ canonicalized for the same reason the shared test does it (a `..` segment or a
3233
+ symlinked component puts a physically-inside path outside lexically), and a host
3234
+ that cannot canonicalize degrades to "inside": the safe direction here is the one
3235
+ that WARNS, matching the shared test's own degrade.
3236
+ """
3237
+ if not task.worktree_path:
3238
+ return False
3239
+ raw = Path(task.spec_file or "")
3240
+ if not raw.is_absolute():
3241
+ return True
3242
+ try:
3243
+ return raw.resolve().is_relative_to(Path(task.worktree_path).resolve())
3244
+ except (OSError, RuntimeError):
3245
+ return True
3246
+
3247
+
3248
+ def redrive_base_ref(state: RunState, *, isolated_redrive: bool) -> str:
3249
+ """The ref whose committed tree the re-drive will actually read this unit's spec
3250
+ from: the run's PINNED `target_branch` when the re-drive will MOUNT, ``HEAD``
3251
+ otherwise.
3252
+
3253
+ Not `HEAD` in both cases, because the isolated re-drive never reads the main
3254
+ checkout's working ref. `engine._finish_inflight` discards the escalated worktree
3255
+ and its branch and `_run_story` mounts a replacement, and
3256
+ `workspace.open_unit_workspace` cuts that fresh branch from the `base` it is handed
3257
+ — `worktree_flow.run_isolated` passes `state.target_branch`, pinned once at run
3258
+ start so resume keeps targeting the same branch. An operator who checks out another
3259
+ branch in the main checkout while the escalation is paused therefore moves `HEAD`
3260
+ off the tree the re-drive reads, in either direction: a correction committed on the
3261
+ now-current branch is invisible to the re-drive, and one committed on the target
3262
+ branch is invisible to `HEAD`.
3263
+
3264
+ That mattered once `rearm-spec-write-unreachable` began holding the resume
3265
+ (`rearm_holds_the_resume`): reading the wrong ref does not merely mis-word a
3266
+ warning, it either resumes a re-drive that re-wedges on the target branch's
3267
+ terminal status, or holds a resume whose work is already committed where the
3268
+ re-drive will find it.
3269
+
3270
+ `isolated_redrive` is the LIVE policy's isolation mode, injected by the caller, and
3271
+ the task drops out of the signature entirely. It used to be inferred from
3272
+ `task.worktree_path` — a recorded mount — and that is the retrospective fact, not
3273
+ this one. `engine._run_story` selects the mode from `self._isolated` alone, and an
3274
+ isolation change mid-run is journalled, never refused, so the recorded mount and the
3275
+ next re-drive part company in BOTH directions: a run flipped to `"none"` still
3276
+ carries the escalated attempt's mount and would name the pinned branch for an
3277
+ in-place re-drive that reads `HEAD`, and one flipped to `"worktree"` carries no
3278
+ mount at all and would name `HEAD` for a re-drive that mounts. Both send a
3279
+ correction to a tree the run does not read. The same injection is how
3280
+ `validate_restore_latch` already learns this fact.
3281
+
3282
+ That the caller must supply it is the point: `froid-loop resolve` computes this
3283
+ context in a SEPARATE process, before the resume ever runs, so no amount of
3284
+ resume-time bookkeeping on `task.worktree_path` could have reached it. The fact
3285
+ enters the pure core as a parameter and nothing here reads policy.
3286
+
3287
+ An empty `target_branch` beside an isolated re-drive is a MISSING value, not a
3288
+ divergent one: `ensure_target_branch` pins the field before any worktree mounts, so
3289
+ only a state.json predating it can reach here, and that shape degrades to exactly
3290
+ the ref it read before — the same migration `restamp_code_root` gives an unrecorded
3291
+ root. Answering ``""`` instead would hold the resume on a per-configuration
3292
+ constant, the failure the record's narrowing exists to avoid.
3293
+ """
3294
+ if isolated_redrive and state.target_branch:
3295
+ return state.target_branch
3296
+ return "HEAD"
3297
+
3298
+
3299
+ def spec_reaches_the_redrive(task: StoryTask, state: RunState, *, isolated_redrive: bool) -> bool:
3300
+ """Whether an edit to this task's spec survives to the re-drive that reads it.
3301
+
3302
+ The other half of `task_spec_path`'s answer, and the two ask different questions of
3303
+ different sources. That one is RETROSPECTIVE — which tree owns the state this task
3304
+ already persisted — and reads the recorded mount, correctly. This one is
3305
+ PROSPECTIVE, so it reads `isolated_redrive`: the live policy's mode, injected by the
3306
+ caller exactly as `redrive_base_ref` and `validate_restore_latch` take it.
3307
+
3308
+ Both arms are about the same gap between where the edit LANDS (`task_spec_path`) and
3309
+ where the re-drive READS:
3310
+
3311
+ - the re-drive will MOUNT: it reads the COMMITTED tree of a fresh worktree, so only
3312
+ a spec outside both checkouts is one file they share
3313
+ (`_spec_is_shared_with_the_redrive` carries that argument in full). True whether
3314
+ or not a mount is recorded — a run flipped to `isolation = "worktree"` mid-pause
3315
+ has none, and its working-tree edit vanishes just as silently.
3316
+ - the re-drive runs IN PLACE: it reads the main checkout's working tree, so the edit
3317
+ reaches unless it landed inside a recorded mount (`_spec_is_inside_the_mount`) —
3318
+ the flip in the other direction.
3319
+
3320
+ Public because `resolve.build_context` needs it for the same reason
3321
+ `rearm_escalation` does: the context hands a human and an agent a `spec_file` to
3322
+ edit, and an edit to a doomed copy is worse than no edit — it looks like it landed.
3323
+ """
3324
+ if isolated_redrive:
3325
+ return _spec_is_shared_with_the_redrive(state, task)
3326
+ return not _spec_is_inside_the_mount(task)
3327
+
3328
+
3329
+ def _upstream_artifacts_folder(state: RunState) -> Path:
3330
+ """The folder holding the UPSTREAM stories artifacts a sentinel's correction goes
3331
+ into — anchored on the project, never on a mount.
3332
+
3333
+ Deliberately NOT `task_stories_root`, which answers "which tree does this RUN read
3334
+ its manifest out of" and is the mount whenever the task holds one. This answers
3335
+ "which folder does the CORRECTION land in", and `resolve.run_session` settles that
3336
+ independently of the mount: the agent runs with `cwd=project` and the artifacts are
3337
+ named by a project-relative `state.spec_folder`, so the writes go to the main
3338
+ checkout even for a task that recorded a worktree. An absolute `spec_folder` — the
3339
+ external-artifact-dir layout `[stories] source` allows — is left where it is, which
3340
+ is what `resolve_spec_folder` already does and what makes it shared across
3341
+ checkouts.
3342
+
3343
+ One locator for all three consumers (the gate, the proof, and the journal record)
3344
+ so a record can never name a folder its own gate did not measure.
3345
+ """
3346
+ from .stories import resolve_spec_folder
3347
+
3348
+ return resolve_spec_folder(Path(state.project), state.spec_folder)
3349
+
3350
+
3351
+ def stories_reach_the_redrive(task: StoryTask, state: RunState, *, isolated_redrive: bool) -> bool:
3352
+ """Whether an edit to this run's UPSTREAM stories artifacts survives to the re-drive.
3353
+
3354
+ `spec_reaches_the_redrive` asked of `SPEC.md` / `stories.yaml` instead of the frozen
3355
+ spec, for the one wedge where the spec is not the artifact being corrected: a
3356
+ fixed-slug pre-planning-halt SENTINEL. A sentinel is cleared by DELETION, so the
3357
+ re-arm drops `task.spec_file` and there is no spec write whose reachability that
3358
+ helper could measure — which is why its arm is an `else` this path never entered,
3359
+ and why no hold ever fired for a sentinel. But the correction that stops the
3360
+ sentinel RECURRING is upstream, in the artifacts `froid-loop-resolve/SKILL.md` sends
3361
+ the agent to instead of the sentinel, and it faces the identical gap: an isolated
3362
+ re-drive mounts fresh from `redrive_base_ref` and re-plans from a COMMITTED tree, so
3363
+ an uncommitted upstream edit is invisible and the re-plan mints the sentinel again.
3364
+
3365
+ The two arms are NOT the spec question's, and the difference is where the write
3366
+ lands. `task_spec_path` re-anchors a spec write ON the recorded mount, so a policy
3367
+ flip separates writer from reader in BOTH directions. The upstream artifacts are
3368
+ named by a project-relative `state.spec_folder` and `resolve.run_session` runs the
3369
+ agent with `cwd=project`, so the correction lands in the MAIN CHECKOUT whichever way
3370
+ the flip went. That collapses one arm:
3371
+
3372
+ - the re-drive runs IN PLACE: it reads the main checkout's working tree —
3373
+ `stories_engine._stories_folder` anchors a relative folder on the live workspace
3374
+ root, which is the project under `isolation = "none"`. Writer and reader are the
3375
+ same tree, so the edit reaches. The recorded mount does not enter into it; a run
3376
+ flipped `"worktree" -> "none"` mid-pause still carries one, and it is not where
3377
+ the correction went.
3378
+ - the re-drive will MOUNT: the fresh worktree is cut from git and checks out TRACKED
3379
+ files, so no working-tree write reaches it — with the single exception
3380
+ `_spec_is_shared_with_the_redrive` carries in full, an artifact dir configured
3381
+ OUTSIDE the project tree, which `ProjectPaths.rebased` leaves exactly where it is
3382
+ and every worktree therefore reads through the same absolute path. True whether or
3383
+ not a mount is recorded: a run flipped `"none" -> "worktree"` has none, and its
3384
+ working-tree edit vanishes just as silently.
3385
+
3386
+ Both roots are tested on the mounting arm for the same reason that helper tests
3387
+ both: worktrees normally sit under `<project>/.froid-loop/runs/`, but
3388
+ `workspace.open_unit_workspace` stores a `.resolve()`d path, so a symlinked
3389
+ `.froid-loop` puts the mount outside the project and "outside the project" alone
3390
+ would not be "shared".
3391
+
3392
+ Canonicalized because a `..` segment or a symlinked component puts a
3393
+ physically-inside path outside lexically, and a host that cannot canonicalize
3394
+ degrades to UNREACHABLE — the direction that warns, matching the degrade both spec
3395
+ helpers already chose.
3396
+ """
3397
+ if not isolated_redrive:
3398
+ return True
3399
+ try:
3400
+ real = _upstream_artifacts_folder(state).resolve()
3401
+ if real.is_relative_to(Path(state.project).resolve()):
3402
+ return False
3403
+ if task.worktree_path and real.is_relative_to(Path(task.worktree_path).resolve()):
3404
+ return False
3405
+ return True
3406
+ except (OSError, RuntimeError):
3407
+ return False
3408
+
3409
+
3410
+ # The two upstream artifacts `froid-loop-resolve/SKILL.md` names for a sentinel wedge:
3411
+ # the epic spec and the story manifest the planner reads. Fixed names, discovered as
3412
+ # siblings in the spec folder (`stories.STORIES_FILENAME`'s own docstring says so), so
3413
+ # the proof below can name them without parsing anything.
3414
+ _UPSTREAM_ARTIFACTS = ("SPEC.md", "stories.yaml")
3415
+
3416
+
3417
+ def _redrive_reads_the_upstream_artifacts(state: RunState) -> bool:
3418
+ """PROOF that the tree the re-drive re-plans from already carries this checkout's
3419
+ upstream artifacts byte-for-byte. ``False`` on every uncertainty.
3420
+
3421
+ `_redrive_spec_status`'s counterpart for the sentinel path, and it exists for the
3422
+ same reason: without it the record its caller writes is a per-configuration
3423
+ CONSTANT. Every isolated stories run resolves its spec folder inside the project,
3424
+ so `stories_reach_the_redrive` answers "unreachable" for 100% of sentinel re-arms
3425
+ under `isolation = "worktree"` — and that record now HOLDS THE RESUME
3426
+ (`rearm_holds_the_resume`), so an unnarrowed gate would not merely train the
3427
+ operator to scroll past a warning, it would turn every one of those re-arms into a
3428
+ two-command gesture for an outcome nothing decided. That is the exact failure the
3429
+ spec arm's own narrowing exists to avoid, and it is worse here.
3430
+
3431
+ There is no status to read for a sentinel — it is cleared by deletion and the
3432
+ re-plan routes on nothing — so the proof is byte equality instead: if the ref the
3433
+ fresh worktree is cut from already holds what this checkout holds, the re-drive
3434
+ re-plans from exactly the tree the operator is looking at and there is nothing left
3435
+ to commit. If it does not, the operator has upstream work the re-drive will not read.
3436
+
3437
+ Read at `redrive_base_ref` and NOT at the code root's `HEAD`, for the reason that
3438
+ function documents: an operator who checks out another branch while the escalation
3439
+ is paused moves `HEAD` off the tree the re-drive reads, in either direction. It is
3440
+ asked for the MOUNTING mode unconditionally, and takes no `isolated_redrive` to say
3441
+ so, because there is exactly one reachable caller and it has already established
3442
+ that: `stories_reach_the_redrive` answers "reaches" for every in-place re-drive, so
3443
+ the `and` short-circuits before this runs. Carrying a second in-place arm here would
3444
+ not be defence in depth — it would SHADOW that one, leaving the reachability arm
3445
+ ungraded by any test and a wrong answer there invisible.
3446
+
3447
+ Every uncertainty answers ``False`` so the record fires and the resume holds: a
3448
+ folder outside the code root (which includes the external artifact dir, already
3449
+ exempted one gate earlier as SHARED), an unreadable working-tree file, an untracked
3450
+ or non-blob path at that ref (the read answers ``None``, which no byte string
3451
+ equals), or any `GitError` — including the project simply not being a repository. Suppression requires proof that the work is already done.
3452
+
3453
+ The blob is materialized through `worktree_file_bytes_at_revision`, not read raw:
3454
+ that function exists for precisely this comparison — a live checkout file against
3455
+ its committed counterpart — because Git's smudge, EOL and working-tree-encoding
3456
+ filters mean a byte-exact LF blob is legitimately a CRLF file on disk under
3457
+ `core.autocrlf=true`. Comparing raw blob bytes would mismatch every artifact on a
3458
+ Windows checkout and re-create, on one platform, the constant this narrowing exists
3459
+ to prevent.
3460
+ """
3461
+ base = _upstream_artifacts_folder(state)
3462
+ code_root = state.code_root
3463
+ ref = redrive_base_ref(state, isolated_redrive=True)
3464
+ for name in _UPSTREAM_ARTIFACTS:
3465
+ live = base / name
3466
+ try:
3467
+ rel = live.relative_to(code_root).as_posix()
3468
+ except ValueError:
3469
+ return False
3470
+ try:
3471
+ committed = verify.worktree_file_bytes_at_revision(code_root, ref, rel)
3472
+ except verify.GitError:
3473
+ return False
3474
+ try:
3475
+ working = live.read_bytes()
3476
+ except OSError:
3477
+ return False
3478
+ if committed != working:
3479
+ return False
3480
+ return True
3481
+
3482
+
3483
+ def _restore_rearmed_spec(
3484
+ spec_path: Path, original: bytes | None, task: StoryTask, state: RunState
3485
+ ) -> None:
3486
+ """Put back the bytes a re-arm FOUND on the spec, for the aborts that can fire after
3487
+ a write has already landed.
3488
+
3489
+ `rearm_escalation` holds an invariant its own refusals depend on: an aborted re-arm
3490
+ leaves the spec byte-identical, so the escalation stays armed and the human can fix
3491
+ the file and re-run resolve. TWO of its four refusals earn that by SEQUENCING alone —
3492
+ the flip's read-back check and the `FrontmatterWriteError` arm both raise before
3493
+ `devcontract.strip_auto_run_result` runs, which is why that strip is deliberately
3494
+ ordered after them, and `set_frontmatter_status` decides it cannot move a `status:`
3495
+ before it writes anything. The other two cannot be sequenced out of the hazard, and
3496
+ both call this:
3497
+
3498
+ * The baseline re-stamp needs `task.baseline_commit` from the advance, and the
3499
+ advance must itself run after the spec block (a just-cleared stories sentinel would
3500
+ otherwise be captured into `baseline_untracked` as phantom pre-existing residue).
3501
+ * The `(OSError, UnicodeDecodeError)` arm spans BOTH spec helpers, and the strip is
3502
+ the later one — a fault raised inside it is raised after the flip published.
3503
+
3504
+ By the time either can fail, the status flip has landed and `save_state` has not — so
3505
+ the abort would otherwise leave the run's task ESCALATED against a spec already
3506
+ flipped to the re-drive's status and (for the re-stamp) stripped of the terminal
3507
+ `## Auto Run Result` the next resolve session reads as its context. That is exactly
3508
+ the "one edit nothing else records" the sequencing exists to prevent.
3509
+
3510
+ Writes only what it can prove it changed. `original` is `None` when the spec was
3511
+ unreadable before the first write (there is then nothing to restore, and nothing
3512
+ could have been written either), and a spec that is gone or unreadable NOW is not a
3513
+ state this undo can improve — recreating a file another process removed would fight
3514
+ a concurrent actor rather than restore this function's own edit. Bytes equal to
3515
+ `original` mean nothing landed, so nothing is rewritten and the mtime is left alone.
3516
+
3517
+ Byte-verbatim and CONFINED, matching the writes it undoes: `atomic_write_text_confined`
3518
+ would re-encode and translate newlines, so a CRLF spec would come back subtly
3519
+ different from the file this re-arm found, and an unconfined write would drop the
3520
+ `O_NOFOLLOW` walk of the parent components (#593) that every other write to this path
3521
+ takes. A restore that itself fails RAISES rather than degrading — the spec is then
3522
+ half-written and only the operator can settle it, which is the loudest thing this can
3523
+ be. `UnconfinedWriteError` is an `OSError`, so the one arm covers both.
3524
+ """
3525
+ if original is None:
3526
+ return
3527
+ try:
3528
+ if spec_path.read_bytes() == original:
3529
+ return
3530
+ except OSError:
3531
+ return
3532
+ try:
3533
+ atomic_write_bytes_confined(spec_path, original, confine_root=task_spec_root(task, state))
3534
+ except OSError as e:
3535
+ raise RearmError(
3536
+ f"cannot restore {spec_path} after a failed re-arm "
3537
+ f"({e.__class__.__name__}: {e}) — the spec carries this re-arm's status flip "
3538
+ "and has lost its `## Auto Run Result` section, while the story is still "
3539
+ "escalated; restore the spec from git, then re-run resolve"
3540
+ ) from e
3541
+
3542
+
3543
+ def _redrive_spec_status(state: RunState, task: StoryTask, *, isolated_redrive: bool) -> str:
3544
+ """The spec's status AS THE RE-DRIVE WILL READ IT, or ``""`` when unprovable.
3545
+
3546
+ The proof that decides whether the operator still has anything to do, so it has to
3547
+ read the same file the caller's remedy names — otherwise the record holds a resume
3548
+ over work that is already done, or clears on work that is not.
3549
+
3550
+ Two sources, because the two re-drive modes read two different things:
3551
+
3552
+ * MOUNTING: the fresh worktree is cut from git and checks out TRACKED files only, so
3553
+ it reads the COMMITTED spec and never a working-tree write. Anchored on
3554
+ `state.code_root` — the same tree the baseline advance reads — at the ref
3555
+ `redrive_base_ref` names, the run's pinned `target_branch` rather than that tree's
3556
+ current `HEAD`.
3557
+ * IN PLACE: the story re-runs in the main checkout, which reads its WORKING TREE. A
3558
+ commit is neither required nor sufficient there, so measuring the committed tree
3559
+ would hold the resume until the operator committed a correction the re-drive would
3560
+ have read uncommitted — and `rearm_event_notice`'s in-place remedy tells them to
3561
+ do exactly that (re-apply it in the main checkout, no commit), so a committed-only
3562
+ proof would make the record's own instruction unable to clear it.
3563
+
3564
+ Reached only when the write does NOT reach the re-drive, so the in-place arm is
3565
+ always the isolation-flip shape: the flip's write landed in the mount the escalated
3566
+ attempt recorded while the re-drive reads `state.project`. That is the tree
3567
+ `task_spec_root` answers for a task with no mount, which is what the resume makes
3568
+ this task once `release_mount_owned_state` runs.
3569
+
3570
+ Degrades to ``""`` on every uncertainty: a spec recorded absolute (nothing names
3571
+ its position in the tree), an absent or non-blob path at that ref, a non-UTF-8 blob,
3572
+ or any `GitError` — which includes the project simply not being a repository, and a
3573
+ `target_branch` the code root no longer carries. ``""`` never equals a target
3574
+ status, so the caller's record still fires. Suppression therefore requires PROOF
3575
+ that the work is already done, and the non-repo case stays non-fatal, as the story's
3576
+ Boundaries require.
3577
+
3578
+ Degrades to ``""`` on every uncertainty in BOTH arms, including a spec recorded
3579
+ absolute. That arm is narrower than it looks: the caller has already answered the one
3580
+ absolute shape whose write the re-drive DOES read — the shared external spec — with
3581
+ `_spec_is_shared_with_the_redrive`. What still reaches here is an absolute spelling
3582
+ of a path inside one of the two checkouts, which is genuinely unreachable, and whose
3583
+ position in the re-drive's tree nothing here can name, so degrading it to a warning
3584
+ is the right answer rather than a gap.
3585
+ """
3586
+ raw = Path(task.spec_file or "")
3587
+ if not task.spec_file or raw.is_absolute():
3588
+ return ""
3589
+ if not isolated_redrive:
3590
+ try:
3591
+ text = (Path(state.project) / raw).read_text(encoding="utf-8")
3592
+ except (OSError, UnicodeDecodeError):
3593
+ return ""
3594
+ return status_of(parse_frontmatter(text))
3595
+ try:
3596
+ blob = verify.file_bytes_at_revision(
3597
+ state.code_root,
3598
+ redrive_base_ref(state, isolated_redrive=isolated_redrive),
3599
+ raw.as_posix(),
3600
+ )
3601
+ except verify.GitError:
3602
+ return ""
3603
+ if blob is None:
3604
+ return ""
3605
+ try:
3606
+ text = blob.decode("utf-8")
3607
+ except UnicodeDecodeError:
3608
+ return ""
3609
+ return status_of(parse_frontmatter(text))
3610
+
3611
+
3612
+ def restamp_code_root(run_dir: Path, repo_root: Path) -> str | None:
3613
+ """Re-point a paused run's persisted code-root mirror at `repo_root` — the tree
3614
+ the caller is about to act in — and return the warning an operator must see when
3615
+ that MOVED a root the run had recorded (`None` when it already agreed, or when the
3616
+ run predates the field).
3617
+
3618
+ Exists because `rearm_escalation` reads that mirror OUT OF PROCESS
3619
+ (`RunState.code_root`) and has no `ProjectPaths` to consult, while `repo_root:` is
3620
+ re-read from config.yaml by every process that arms an engine. `cli._resume_paused_run`
3621
+ folds the same re-stamp into the one `save_state` that also carries the policy
3622
+ snapshot and the config digest — this is the seam for the surfaces that re-arm
3623
+ BEFORE they resume (`cli.cmd_resolve`, `TuiApp._do_rearm`), where that write lands
3624
+ too late to aim the re-arm.
3625
+
3626
+ The compare is exact and uncanonicalized, matching resume's: both sides are
3627
+ `str(paths.repo_root)` off `froidconfig.load_paths`, which resolves every member or
3628
+ raises, so they are spelled the same way whenever they name the same tree. An empty
3629
+ recorded root is a MISSING value, not a divergent one — a state.json written before
3630
+ the field existed — so it is migrated silently and reported as no move.
3631
+
3632
+ The message names neither tree, like resume's: what an operator needs is that the
3633
+ run has changed repositories, and the paths are the half that would put an
3634
+ attacker-controlled string on their terminal.
3635
+ """
3636
+ state = load_state(run_dir)
3637
+ new = str(repo_root)
3638
+ if state.repo_root == new:
3639
+ return None
3640
+ moved = bool(state.repo_root)
3641
+ state.repo_root = new
3642
+ save_state(run_dir, state)
3643
+ if not moved:
3644
+ return None
3645
+ return (
3646
+ f"run {run_dir.name}: the code root in _froid/bmm/config.yaml has changed since "
3647
+ "this run started — the re-drive works in the tree configured now, while the "
3648
+ "baselines, preserve refs and branches this run already recorded name objects "
3649
+ "in the previous one. Restore the previous `repo_root:` value if you did not "
3650
+ "intend the move."
3651
+ )
3652
+
3653
+
3654
+ def rearm_escalation(
3655
+ run_dir: Path,
3656
+ story_key: str | None = None,
3657
+ *,
3658
+ restore_patch: str | None = None,
3659
+ isolated_redrive: bool,
3660
+ ) -> str:
3661
+ """Re-arm an escalation-paused story so the next resume re-drives it.
3662
+
3663
+ Flips the escalated task out of its terminal ESCALATED phase back to
3664
+ PENDING — which makes `_finish_inflight` reset the tree to the story's
3665
+ baseline and re-run it (clean rebuild) against the now-corrected frozen
3666
+ spec. The baseline itself is advanced to the CODE TREE's current HEAD
3667
+ (`state.code_root`, which is `paths.repo_root` — the tree the dev writer
3668
+ stamps from and the proof-of-work gate measures, and the same directory as
3669
+ `state.project` in every configuration without a `repo_root:` override) and
3670
+ the untracked snapshot refreshed, so commits and files the resolve session
3671
+ produced count as the rebuild's starting point, not as attempt debris to
3672
+ roll back. Strips the escalated attempt's stale `## Auto Run Result`
3673
+ section so the re-drive cannot read as terminal from its first save, and
3674
+ sets the spec's frontmatter status so step-01 routes to the right stage.
3675
+ Does NOT clear the pause; the caller resumes the run separately.
3676
+
3677
+ Two consequences of the reset are load-bearing and easy to undo by accident:
3678
+
3679
+ - `task.generation` is bumped, because `attempt` returning to 0 would
3680
+ otherwise let the re-drive re-mint a session id byte-equal to one the
3681
+ abandoned attempt already recorded (#705). `task.sessions` is deliberately
3682
+ NOT cleared — a second resolve cycle reads that run-dir audit trail — so
3683
+ the id is what has to change.
3684
+ - The spec's `baseline_revision` is re-stamped on BOTH legs, and only when the
3685
+ advance above actually RAN — `advanced` records that both git reads succeeded,
3686
+ not that HEAD changed, so a resolve session that committed nothing still
3687
+ re-stamps (with the same sha, harmlessly). What it will not do is re-stamp
3688
+ after a FAILED advance (see the block that does it for why each half of that
3689
+ is the way it is).
3690
+
3691
+ Two re-drive modes, selected by `restore_patch`:
3692
+
3693
+ - **from-scratch** (default, ``restore_patch=None``): status → ``ready-for-dev``
3694
+ so the dev session re-implements from a clean baseline. Assigning None also
3695
+ clears any stale latch from a prior restore attempt the human abandoned.
3696
+ - **patch-restore** (Froid Plane #2564, ``restore_patch`` set): the human
3697
+ confirmed the escalated attempt's reading was correct. Status → ``in-review``
3698
+ so step-01 routes straight to step-04, and the path is latched onto the task
3699
+ (`task.restore_patch`) so the engine re-applies the saved patch onto the
3700
+ baseline before dispatching — the re-driven session resumes review on the
3701
+ restored diff instead of re-implementing. The status is set here
3702
+ deterministically; the resolve agent must NOT set it. Because the baseline
3703
+ advances (above) while the patch was diffed from the OLD baseline, a resolve
3704
+ session that committed changes to the patched files makes the re-drive's
3705
+ apply fail — the engine then escalates loudly instead of dispatching on a
3706
+ half-restored tree (see verify.apply_patch).
3707
+
3708
+ Stories mode: when the escalated spec is a fixed-slug sentinel
3709
+ (`<id>-unresolved.md` / `<id>-ambiguous.md`, written by a pre-planning HALT),
3710
+ it cannot be re-opened by a status flip — its very presence wedges the id.
3711
+ Instead preserve a copy under `{run_dir}/sentinels/`, journal `sentinel-cleared`
3712
+ with the blocking condition, and delete it, so the re-dispatch resolves to a
3713
+ clean PENDING and re-plans from scratch (leg 1 again for a spec_checkpoint id).
3714
+
3715
+ `isolated_redrive` is the LIVE policy's isolation mode (`scm.isolation ==
3716
+ "worktree"`), which run state cannot carry: the mode is re-read at every resume and
3717
+ a mid-run change is journalled, never refused, so the recorded `task.worktree_path`
3718
+ says how the escalated attempt RAN and only policy says how the re-drive WILL run.
3719
+ Keyword-only and required, because every consumer of it here is an answer a human
3720
+ acts on — which ref to commit the corrected spec on, whether the working-tree flip
3721
+ reaches the re-drive at all, whether a restore latch can be honored — and a
3722
+ defaulted mode would answer all three for the wrong tree in silence, which is the
3723
+ defect this parameter exists to close. Both callers (`cli.cmd_resolve`,
3724
+ `tui.TuiApp._do_rearm`) hold a loaded policy already.
3725
+
3726
+ Returns the re-armed story key. Raises RearmError when the run is not paused at
3727
+ the escalation stage, the target story is not escalated, or a supplied
3728
+ `restore_patch` fails `validate_restore_latch` (the shared precondition set —
3729
+ sentinel wedge, spec-less escalation, worktree isolation).
3730
+ """
3731
+ state = load_state(run_dir)
3732
+ if state.paused_stage != PAUSE_ESCALATION:
3733
+ raise RearmError(
3734
+ f"run {run_dir.name} is not paused at an escalation "
3735
+ f"(stage: {state.paused_stage or 'none'})"
3736
+ )
3737
+ key = story_key or state.paused_story_key
3738
+ if key is None:
3739
+ raise RearmError(f"run {run_dir.name} has no escalated story to resolve")
3740
+ task = state.tasks.get(key)
3741
+ if task is None:
3742
+ raise RearmError(f"run {run_dir.name} has no task for story {key}")
3743
+ if task.phase != Phase.ESCALATED:
3744
+ raise RearmError(f"story {key} is not escalated (phase: {task.phase})")
3745
+ # Patch-restore preconditions (T1 guard + spec-less wedge + worktree isolation),
3746
+ # rejected here before any task mutation so the escalation stays armed for a
3747
+ # corrected resolve. `cli._resolve_restore_patch` runs the same validator ahead
3748
+ # of the interactive session; this call is what makes a programmatic caller
3749
+ # (TUI restore parity, scripts) unable to bypass it.
3750
+ if restore_patch:
3751
+ err = validate_restore_latch(state, task, key, worktree_isolation=isolated_redrive)
3752
+ if err is not None:
3753
+ raise RearmError(err)
3754
+
3755
+ journal = Journal(run_dir)
3756
+ # Read before the unconditional overwrite below: they describe the restore
3757
+ # attempt this re-arm is abandoning, and the residue block needs both.
3758
+ old_latch = task.restore_patch
3759
+ old_baseline = task.baseline_commit
3760
+ # deliberate reset, not a normal state-machine transition (mirrors
3761
+ # engine._finish_inflight): a clean re-attempt against the corrected spec.
3762
+ task.phase = Phase.PENDING
3763
+ task.attempt = 0
3764
+ # A new generation of this task. `attempt` going back to 0 (and the next
3765
+ # dispatch bumping it to 1) would otherwise re-mint a session task_id
3766
+ # byte-equal to one the abandoned attempt already recorded, and
3767
+ # `Engine._resumable_session` — matching that id over the append-only
3768
+ # `task.sessions`, which this function deliberately does NOT clear — would
3769
+ # replay the abandoned verdict for the fresh attempt (#705). Bumped BEFORE any
3770
+ # dispatch, so the id is unique from the re-drive's first session onward.
3771
+ task.generation += 1
3772
+ task.review_cycle = 0
3773
+ task.followup_reviews_spent = 0 # human-resolved re-drive gets a fresh damping budget
3774
+ task.defer_reason = None
3775
+ task.rearmed = True # resume-time recovery notice describes a clean rebuild,
3776
+ # not a failed attempt (engine._finish_inflight clears it once the rebuild runs)
3777
+ # Always (re)assign the latch: a None restore_patch clears a stale one left by
3778
+ # a prior restore attempt the human then chose to redo from scratch.
3779
+ task.restore_patch = restore_patch
3780
+
3781
+ # The bytes this re-arm found on the spec, for `_restore_rearmed_spec`. Declared out
3782
+ # here because the baseline re-stamp that consumes it sits in a SECOND
3783
+ # `if task.spec_file:` block, past the advance it depends on.
3784
+ spec_before: bytes | None = None
3785
+ if task.spec_file:
3786
+ spec_path = task_spec_path(task, state)
3787
+ # Stories mode only: a fixed-slug pre-planning-halt sentinel
3788
+ # (`<id>-unresolved.md` / `<id>-ambiguous.md`) is cleared by deletion, not a
3789
+ # status flip. Clear it ONLY when the run recorded this task AS a sentinel at
3790
+ # detection time (`task.sentinel_kind`, stamped by StoriesEngine's pick-time
3791
+ # wedge / post-dev read-back) — never by re-deriving from the basename. That
3792
+ # keeps a real story spec that merely happens to be named `<key>-unresolved.md`,
3793
+ # or a *non-sentinel* escalation whose spec matches the convention, on the
3794
+ # status-flip path so it is kept, not deleted. Gate on the run source too (the
3795
+ # convention exists only in stories mode) and defensively re-confirm the
3796
+ # on-disk name still matches the recorded slug before deleting.
3797
+ sentinel_kind = task.sentinel_kind if state.source == "stories" else ""
3798
+ if sentinel_kind and _sentinel_condition(spec_path, key) == sentinel_kind:
3799
+ # a sentinel is cleared by deletion, not a status flip; drop the stale
3800
+ # spec_file so the re-dispatch starts from PENDING (clean re-plan).
3801
+ _clear_sentinel(run_dir, journal, spec_path, key, sentinel_kind)
3802
+ task.spec_file = None
3803
+ task.sentinel_kind = "" # verdict discharged; the re-dispatch is clean
3804
+ # Deleting the sentinel does not make the re-plan produce a different one:
3805
+ # the correction that does lives UPSTREAM, in the `SPEC.md` / `stories.yaml`
3806
+ # the resolve skill sends the agent to instead of this file. That correction
3807
+ # faces the same reachability gap the spec arm below measures, and faced NO
3808
+ # gate at all — this arm cleared `spec_file` and fell through, so
3809
+ # `write_reaches_the_redrive` was never computed and the resume was never
3810
+ # held for a sentinel. An isolated re-drive then mounts fresh from
3811
+ # `redrive_base_ref`, re-plans from a committed tree that never saw the
3812
+ # edit, mints the same sentinel again, and the escalation is spent.
3813
+ #
3814
+ # Narrowed by PROOF for the reason the spec record below is, and the need is
3815
+ # sharper here: `stories_reach_the_redrive` answers "unreachable" for EVERY
3816
+ # isolated stories run whose spec folder sits inside the project, which is
3817
+ # every one we author. Gating on it alone would fire — and hold the resume —
3818
+ # on 100% of isolated sentinel re-arms, a per-configuration constant rather
3819
+ # than an event. `_redrive_reads_the_upstream_artifacts` is what makes it an
3820
+ # event: it fires only while this checkout still holds upstream bytes the
3821
+ # ref the re-drive mounts from does not.
3822
+ #
3823
+ # No `redrive` discriminator, unlike the spec record: this one has a single
3824
+ # remedy because it has a single reachable shape. An in-place re-drive reads
3825
+ # the main checkout's working tree, which is exactly where `cwd=project` put
3826
+ # the correction, so `stories_reach_the_redrive` short-circuits that leg to
3827
+ # reachable and no record is written for it at all.
3828
+ if not stories_reach_the_redrive(
3829
+ task, state, isolated_redrive=isolated_redrive
3830
+ ) and not _redrive_reads_the_upstream_artifacts(state):
3831
+ journal.append(
3832
+ "rearm-upstream-write-unreachable",
3833
+ story_key=key,
3834
+ # `task_stories_root` names the tree the RUN owns; the correction
3835
+ # lands in the checkout the resolve session ran in. Both are the
3836
+ # project on this leg unless a mount is recorded, and the operator
3837
+ # needs the folder to act, so the record carries the folder the
3838
+ # remedy is about rather than the run's read locator.
3839
+ stories_root=str(_upstream_artifacts_folder(state)),
3840
+ target_branch=state.target_branch,
3841
+ )
3842
+ else:
3843
+ # A WORKTREE-LOCAL spec's writes below land in the unit's worktree
3844
+ # (`task_spec_path`) — which the re-drive destroys before reading anything.
3845
+ # A re-armed task (phase PENDING, `defer_reason` cleared, and no resumable
3846
+ # session because `generation` was just bumped) falls to
3847
+ # `engine._finish_inflight`'s final arm, which calls `discard_worktree` and
3848
+ # lets `_run_story` mount a fresh one. The re-driven session then resolves
3849
+ # its spec through `verify.resolve_spec_path(task.spec_file,
3850
+ # workspace.paths)` (`engine._dispatched_spec_for_attempt`), and under
3851
+ # isolation `workspace.paths` is rebased onto that FRESH worktree, which
3852
+ # checks out TRACKED files only. So the re-drive reads the COMMITTED spec.
3853
+ #
3854
+ # No working-tree write reaches it — not this one, and not a write to the
3855
+ # main checkout either: the fresh worktree comes from git rather than from a
3856
+ # copy of that tree, and `seed_adapter_defaults` seeds adapter config files,
3857
+ # not the output folder. The channel that DOES work is the human committing
3858
+ # the corrected spec from the resolve session, which runs with `cwd=project`.
3859
+ # The writes below are kept (they are correct for the in-place case, and
3860
+ # harmless here), but the operator is told — a flip that cannot land is
3861
+ # exactly the silent re-wedge #640(b) exists to end.
3862
+ #
3863
+ # "Worktree-local" is the load-bearing qualifier, and isolation does not
3864
+ # imply it: an artifact dir configured OUTSIDE the project tree is shared
3865
+ # across checkouts by `ProjectPaths.rebased`, so a spec that landed there is
3866
+ # one file the fresh worktree reads through the very absolute path this
3867
+ # writes to. `_spec_is_shared_with_the_redrive` carves out that case, and only
3868
+ # that one: the main checkout's copy is outside the worktree too, and stays
3869
+ # unreachable because the re-drive measures it against worktree-local roots.
3870
+ # Route /froid-build-auto via the spec's frontmatter status (decision
3871
+ # table): patch-restore -> in-review -> step-04 (resume review on
3872
+ # the restored diff); from-scratch -> ready-for-dev -> step-03
3873
+ # (re-implement). Independent of the resolve agent having set it.
3874
+ target_status = "in-review" if restore_patch else "ready-for-dev"
3875
+ # Whether the writes below are the copy the re-driven session actually
3876
+ # reads. Hoisted out of the record's condition because TWO decisions turn on
3877
+ # it, and only one of them used to: the warning below, and the flip's
3878
+ # REFUSAL one screen down, which was gated on `spec_path.is_file()` alone.
3879
+ # Under isolation that readable file is the doomed worktree copy, so the
3880
+ # refusal demanded a repair to the one file the re-drive destroys before
3881
+ # reading anything — and demanded it even when `_redrive_spec_status` had
3882
+ # already proven the committed spec carries the status the re-drive routes
3883
+ # on. See `_spec_is_shared_with_the_redrive` for why an isolated unit's spec
3884
+ # is nevertheless reachable when it sits in an artifact dir configured
3885
+ # outside the project tree.
3886
+ write_reaches_the_redrive = spec_reaches_the_redrive(
3887
+ task, state, isolated_redrive=isolated_redrive
3888
+ )
3889
+ # Narrowed to the case an operator can ACT on. Every isolated escalation
3890
+ # carries a mounted `worktree_path` — `worktree_flow.escalate_unit` never
3891
+ # clears it, and `keep_branch_and_escalate` deliberately leaves the worktree
3892
+ # up — so gating on that alone fired this warning on 100% of re-arms under
3893
+ # `isolation = "worktree"`: a per-configuration constant, not an event, and
3894
+ # the same "trains the operator to scroll past the meaningful one" failure
3895
+ # that the `flipped` read-back below and the `overwritten != old_baseline`
3896
+ # guard were each narrowed to avoid. The remedy it prints ("commit the
3897
+ # corrected spec") is already a no-op once the committed spec carries the
3898
+ # target status, which is precisely when the re-drive reads what it needs.
3899
+ # Suppression requires PROOF: an unreadable blob, a non-repo project, or any
3900
+ # git fault leaves `""` and the record fires. The proof is read at
3901
+ # `redrive_base_ref`, NOT at the code root's current `HEAD` — the two part
3902
+ # company as soon as the operator checks out another branch while the
3903
+ # escalation is paused, and this record now holds the resume.
3904
+ #
3905
+ # The branch rides along because the remedy needs it: on exactly the shape
3906
+ # the ref fix rescues, "commit the corrected spec" without a branch sends
3907
+ # the operator to commit again on the branch the re-drive does not read, and
3908
+ # the next re-arm prints the same sentence. Empty for the migrated shape
3909
+ # `redrive_base_ref` degrades to `HEAD` for, and the notice drops the
3910
+ # clause rather than naming a ref it cannot source — and empty for an
3911
+ # IN-PLACE re-drive, which has no branch to name at all.
3912
+ #
3913
+ # `redrive` is that second shape's discriminator, and it goes ON the record
3914
+ # because the reader is out of process: `rearm_event_notice` renders from a
3915
+ # journal line alone and cannot re-read the policy that produced it. One
3916
+ # kind, two remedies. Isolated: the writes landed in a mount the re-drive
3917
+ # discards, so the correction must be COMMITTED on the named branch. In
3918
+ # place: the writes landed in the mount the escalated attempt recorded while
3919
+ # the re-drive now reads the main checkout, so the correction must be made
3920
+ # THERE — a commit is neither required nor sufficient. Telling the second
3921
+ # operator to commit sends them to the wrong tree, which is the same class
3922
+ # of silent loss this whole record exists to end.
3923
+ #
3924
+ # Spelled `target_branch` and NOT `base`, because `diagnostics` routes the
3925
+ # scrub by field NAME: `target_branch` is already in `_JOURNAL_ALIAS_FIELDS`
3926
+ # under the `branch` namespace (with no journal producer until now), while
3927
+ # any new spelling falls through to `scrub_json`, which waves an
3928
+ # identifier-shaped branch name through verbatim. In a normal run
3929
+ # `ensure_target_branch` has already journalled the same string as `branch`,
3930
+ # so the egress backstop would repair it and disclose a `backstop_repairs`
3931
+ # routing gap; in a truncated journal missing that event nothing would catch
3932
+ # it and the branch would ship in a shareable bundle. `target` — the
3933
+ # spelling the merge kinds use — is NOT available: `board-advance-*` puts a
3934
+ # sprint STATUS in that same field, and routing is by name, so aliasing it
3935
+ # to `branch` would pseudonymize statuses as branches.
3936
+ if (
3937
+ not write_reaches_the_redrive
3938
+ and _redrive_spec_status(state, task, isolated_redrive=isolated_redrive)
3939
+ != target_status
3940
+ ):
3941
+ journal.append(
3942
+ "rearm-spec-write-unreachable",
3943
+ story_key=key,
3944
+ spec_file=str(spec_path),
3945
+ status=target_status,
3946
+ target_branch=state.target_branch if isolated_redrive else "",
3947
+ redrive="isolated" if isolated_redrive else "in-place",
3948
+ )
3949
+ # Captured immediately before the FIRST write, so an abort further down can
3950
+ # put the spec back exactly as found. Unreadable degrades to `None`: the
3951
+ # writes below answer such a path with `False` rather than an exception, so
3952
+ # there would be nothing to undo either.
3953
+ try:
3954
+ spec_before = spec_path.read_bytes()
3955
+ except OSError:
3956
+ spec_before = None
3957
+ try:
3958
+ flipped = verify.set_frontmatter_status(
3959
+ spec_path, target_status, confine_root=task_spec_root(task, state)
3960
+ )
3961
+ # `set_frontmatter_status` answers "nothing to change" with `False`
3962
+ # for FOUR causes, not three — its own docstring lists them: no file,
3963
+ # no frontmatter block, no top-level `status:`, and ALREADY AT THE
3964
+ # TARGET (`_edit_frontmatter_block` returns None on
3965
+ # `original[key] == value`). Only the first three are failures. The
3966
+ # fourth is an ordinary, fully-successful re-arm: a second resolve
3967
+ # cycle on an already-flipped spec, or the documented
3968
+ # `resolve --no-interactive` flow where a human fixed the spec
3969
+ # themselves — the case the comment above calls "Independent of the
3970
+ # resolve agent having set it". Journalling it fired the operator
3971
+ # warning ("could not be re-opened … may re-wedge on it") on a spec
3972
+ # that was byte-identical and CORRECT, which is the "trains the
3973
+ # operator to scroll past the meaningful one" failure the re-stamp's
3974
+ # `overwritten != old_baseline` guard exists to prevent one screen
3975
+ # below. Read the status back to tell the two apart: `read_frontmatter`
3976
+ # degrades a missing/unreadable/unparseable spec to `{}` and `status_of`
3977
+ # then answers `""`, so all three real failures still record.
3978
+ if not flipped and verify.status_of(verify.read_frontmatter(spec_path)) != (
3979
+ target_status
3980
+ ):
3981
+ # Discarding that return is how the flip
3982
+ # became a SILENT no-op: the re-drive is dispatched anyway, step-01
3983
+ # reads the unchanged terminal status, routes the session to "ingest
3984
+ # as context, do not resume", and the story re-wedges with nothing on
3985
+ # the record. The `FrontmatterWriteError` arm below covers only the
3986
+ # shapes that RAISE; this covers the ones that lie quietly.
3987
+ # `refused` is written ON the record because ONE kind now covers
3988
+ # two outcomes and the operator surfaces must tell them apart —
3989
+ # they read the journal OUT OF PROCESS, with neither the task nor
3990
+ # the tree to re-derive it from. Printing the refusal's remedy
3991
+ # ("add a top-level `status:`") for a re-arm that COMPLETED sends
3992
+ # the human to repair a file nothing will read.
3993
+ refused = spec_path.is_file() and write_reaches_the_redrive
3994
+ journal.append(
3995
+ "rearm-spec-flip-skipped",
3996
+ story_key=key,
3997
+ spec_file=str(spec_path),
3998
+ status=target_status,
3999
+ refused=refused,
4000
+ )
4001
+ # ...and then ABORT — but only for a spec that IS a readable file
4002
+ # here AND is the copy the re-drive reads. The first half is the same
4003
+ # `is_file` split the baseline re-stamp below already draws, and for
4004
+ # the same reason. On THAT shape the failure is
4005
+ # a REPAIR that did not land on the very file the re-drive reads, so it
4006
+ # aborts for the same reason the `FrontmatterWriteError` arm does:
4007
+ # journalling alone left the two default surfaces telling the operator
4008
+ # "re-armed <story>" and resuming in the same gesture, so the record's
4009
+ # own imperative was already unactionable when it rendered — while
4010
+ # step-01's contract for what reaches here is not a maybe. A spec with
4011
+ # no `status:` HALTs blocked on `unrecognized status in existing story
4012
+ # file`; one still carrying the escalated attempt's terminal status
4013
+ # routes to "ingest as context, do not resume". Either way the re-drive
4014
+ # re-wedges and the escalation is burned. Refusing keeps it armed: nothing
4015
+ # is persisted yet (`save_state` runs below), the spec is byte-identical
4016
+ # (the `## Auto Run Result` strip is deliberately sequenced AFTER this
4017
+ # check so an abort leaves nothing half-done), and the human fixes the
4018
+ # frontmatter and re-runs resolve.
4019
+ #
4020
+ # A spec that is NOT a file from here keeps warn-and-continue, because
4021
+ # there the flip's failure says nothing about what the re-drive will
4022
+ # read: `spec_file` is persisted RELATIVE to a worktree, an isolated
4023
+ # task's worktree may already be gone, and the re-drive mounts a fresh
4024
+ # one and reads the COMMITTED spec regardless. Aborting on it would
4025
+ # refuse the re-arms that the `rearm-baseline-restamp-skipped` and
4026
+ # `rearm-spec-write-unreachable` records exist to report rather than
4027
+ # prevent — an unreadable path is an observation, and observations
4028
+ # degrade.
4029
+ #
4030
+ # A worktree-local spec that IS readable takes that same lane, for a
4031
+ # sharper version of the same reason: `task_spec_root` anchors this
4032
+ # write on the mounted worktree, so the readable file is the copy the
4033
+ # re-drive DISCARDS. The refusal's own remedy could not fix anything
4034
+ # there — an operator who added a `status:` to that file and re-ran
4035
+ # resolve would flip a spec that is deleted before it is read, while
4036
+ # the committed spec, the one thing that decides routing, went
4037
+ # untouched. Worse, the refusal fired even when the correction was
4038
+ # already committed: `_redrive_spec_status` had just PROVEN the
4039
+ # re-drive routes correctly, and the re-arm was refused anyway over an
4040
+ # obsolete copy. The real remedy on that shape is
4041
+ # `rearm-spec-write-unreachable`'s ("commit the corrected spec"),
4042
+ # which fires from the block above on exactly the legs that need it
4043
+ # and now holds the resume rather than merely printing.
4044
+ #
4045
+ # The record is written on BOTH sides of that split: the abort message
4046
+ # reaches stderr only, and the journal is the run's audit trail —
4047
+ # `_echo_rearm_events` surfaces it from a `finally` on this path.
4048
+ if refused:
4049
+ raise RearmError(
4050
+ f"cannot re-open story spec {spec_path} to `{target_status}` "
4051
+ "for the re-drive: it has no frontmatter `status:` this re-arm "
4052
+ "can set, so the re-driven session would wedge on the status "
4053
+ "it reads — add a top-level `status:` to the spec's "
4054
+ "frontmatter block, then re-run resolve"
4055
+ )
4056
+ # drop the stale `## Auto Run Result` section along with the status flip
4057
+ # (mirrors engine._reset_spec_for_repair): find_result_artifact keys on
4058
+ # that heading, so leaving it would let the re-driven session's first
4059
+ # save of the spec parse as the prior attempt's terminal outcome.
4060
+ #
4061
+ # Sequenced AFTER the read-back check above, not with the flip it mirrors:
4062
+ # that check now raises, and an aborted re-arm must leave the spec exactly
4063
+ # as it found it — a stripped result section on a spec the re-arm then
4064
+ # refused would be the one edit nothing else records.
4065
+ devcontract.strip_auto_run_result(
4066
+ spec_path, confine_root=task_spec_root(task, state)
4067
+ )
4068
+ except verify.FrontmatterWriteError as e:
4069
+ # The spec reads fine but carries `status:` in a shape no line
4070
+ # edit can move (a block scalar, a flow mapping, a value continued
4071
+ # on the next line). This used to be a silent no-op on a bool
4072
+ # nobody read: the re-drive was dispatched anyway, step-01 saw the
4073
+ # unchanged terminal status and routed the session to "ingest as
4074
+ # context, do not resume", and the story re-wedged with nothing on
4075
+ # the record explaining why. Abort here for the same reason as
4076
+ # below, with the remedy this cause actually has.
4077
+ raise RearmError(
4078
+ f"cannot re-open story spec {spec_path} for the re-drive: {e} "
4079
+ f"— the re-drive would repeat the wedge it is meant to clear"
4080
+ ) from e
4081
+ except (OSError, UnicodeDecodeError) as e:
4082
+ # Both helpers re-read the spec as UTF-8; an undecodable PRESENT
4083
+ # spec is a first-class escalation state (resolve_story_spec
4084
+ # degrades it to a wedge), so it can reach this flip. Without the
4085
+ # flip the re-drive would just re-wedge — abort BEFORE any state
4086
+ # is persisted (save_state runs below) with an actionable error
4087
+ # instead of a traceback; the escalation stays armed for a retry.
4088
+ #
4089
+ # ...and this arm is the SECOND refusal that can fire after a write has
4090
+ # landed, which the sequencing argument above does not cover. It guards
4091
+ # BOTH helpers, and `strip_auto_run_result` is the later one: by the
4092
+ # time its own read/decode or its atomic write faults (an
4093
+ # `atomic_write_bytes_confined` that cannot land — ENOSPC, EIO, a
4094
+ # component swapped for a link under the `O_NOFOLLOW` walk — or a spec
4095
+ # replaced under us between the two writes), the flip has already been
4096
+ # published and `save_state` has not. Ordering the strip after the
4097
+ # read-back check bought that check its byte-identical abort; it buys
4098
+ # this one nothing, because the fault is IN the strip. So the same undo
4099
+ # the re-stamp carries applies here, on the same terms.
4100
+ #
4101
+ # On the arm's other shape — the flip itself faulting on an
4102
+ # unreadable/undecodable spec — nothing was written, `spec_before` still
4103
+ # equals the bytes on disk, and `_restore_rearmed_spec` proves that and
4104
+ # returns without touching the file or its mtime.
4105
+ _restore_rearmed_spec(spec_path, spec_before, task, state)
4106
+ raise RearmError(
4107
+ f"cannot re-open story spec {spec_path} for the re-drive "
4108
+ f"({e.__class__.__name__}: {e}) — fix or replace the file "
4109
+ f"(it must be readable UTF-8), then re-run resolve"
4110
+ ) from e
4111
+
4112
+ # A previous restore latch is being replaced (or re-latched onto the same
4113
+ # patch): the abandoned attempt applied that patch, so its NEW files sit
4114
+ # untracked in the tree right now. The refresh below would capture them as
4115
+ # "pre-existing" — after which every rollback preserves them and
4116
+ # finalize_commit's `add -A` sweeps the abandoned attempt into the corrected
4117
+ # story's commit. Subtract them instead (issue #90).
4118
+ #
4119
+ # Runs after the spec block for the same reason the refresh does (a cleared
4120
+ # sentinel must not be snapshotted), and before it because it feeds it.
4121
+ # Nothing is deleted here: the re-drive's reset (verify.safe_rollback) removes
4122
+ # whatever the refreshed snapshot no longer blesses, at the right moment.
4123
+ # The CODE tree, not `state.project`: every git read below (and every baseline
4124
+ # the proof-of-work gate later measures against) must name the repository the
4125
+ # dev writer stamps.
4126
+ #
4127
+ # That is `paths.repo_root` for every run this function can be reached from, but
4128
+ # NOT because `paths.repo_root == workspace.root` universally — it does not.
4129
+ # `Workspace.default` sets `root=paths.repo_root`, while the isolation constructor
4130
+ # mounts `root=<run_dir>/worktrees/<unit>` and rebases a fresh `ProjectPaths` onto
4131
+ # it, so under `isolation = "worktree"` the run-level `repo_root` is the main
4132
+ # checkout and the baseline is stamped in the worktree.
4133
+ #
4134
+ # `froidconfig.worktree_isolation_conflict` refuses worktree isolation beside a
4135
+ # `repo_root:` OVERRIDE — a narrower fact than it looks. It forces
4136
+ # `repo_root == project`; it says nothing about `repo_root` vs `workspace.root`.
4137
+ # Under plain isolation with NO override those two still diverge and isolation is
4138
+ # ON, so "wherever the roots could diverge, isolation is off" is false, and a rule
4139
+ # built on it licenses treating `state.code_root` as the tree the dev writer
4140
+ # stamped — which under isolation it is not.
4141
+ #
4142
+ # What is true, and the only claim to carry forward: `repo_root == project` in
4143
+ # every reachable configuration, so reading HEAD here is right for the in-place
4144
+ # case; and under isolation this value is deliberately SUPERSEDED rather than
4145
+ # relied on — `engine._finish_inflight` discards the worktree and `_dev_phase`
4146
+ # re-stamps `task.baseline_commit` from the fresh worktree's HEAD before any gate
4147
+ # reads it. Do not carry an identity into new code; carry this argument.
4148
+ #
4149
+ # A pre-upgrade state.json with no recorded root degrades to `project` exactly as
4150
+ # before.
4151
+ repo = state.code_root
4152
+ stale_residue = _stale_restore_residue(repo, journal, key, old_latch, old_baseline)
4153
+
4154
+ # Advance the attempt baseline to the CODE TREE's current HEAD (`repo`, above)
4155
+ # and refresh the untracked snapshot: whatever the human-driven resolve session left on the
4156
+ # branch (a committed fixture, a corrected ledger, ...) is authorized input
4157
+ # for the re-drive, not failed-attempt debris. Without this, the re-drive's
4158
+ # reset-to-baseline in engine._rollback_or_pause parks the resolution
4159
+ # commits on an attempt-preserve ref and rebuilds against a tree that
4160
+ # contradicts the corrected spec — the re-driven dev session then hits the
4161
+ # very gap the human just resolved. Best-effort: on a git failure the old
4162
+ # baseline stands (the redrive rollback path tolerates a stale baseline; it
4163
+ # just loses this protection).
4164
+ # Runs AFTER the spec block so a just-cleared stories sentinel (an untracked
4165
+ # file removed above) is not captured into baseline_untracked as a phantom
4166
+ # pre-existing untracked file. The two locals are computed before either task
4167
+ # field is assigned, so a failure on either git call can't advance
4168
+ # baseline_commit while baseline_untracked stays stale, or vice versa.
4169
+ advanced = False
4170
+ try:
4171
+ head = verify.rev_parse_head(repo)
4172
+ untracked = sorted(verify.untracked_files(repo) - stale_residue)
4173
+ except verify.GitError as e:
4174
+ # `verify.GitError` is a TOTAL replacement for the `except Exception` that
4175
+ # stood here, not a narrowing that leaks: both calls go through `_run_git`,
4176
+ # which translates spawn (`GitSpawnError`), timeout (`GitTimeoutError`) and
4177
+ # decode faults into this one taxonomy, and a non-zero rc into a plain
4178
+ # `GitError`. Still swallowed rather than raised — a project that is not a
4179
+ # git repo must not fail re-arm — but no longer SILENT: the degrade is the
4180
+ # difference between "the re-drive starts from the resolution" and "it
4181
+ # rebuilds against the tree the human just corrected away", and the
4182
+ # re-stamp below now refuses to paper over it.
4183
+ journal.append(
4184
+ "rearm-baseline-advance-failed",
4185
+ story_key=key,
4186
+ repo=str(repo),
4187
+ baseline=old_baseline or "",
4188
+ error=f"{e.__class__.__name__}: {e}",
4189
+ )
4190
+ else:
4191
+ task.baseline_commit = head
4192
+ task.baseline_untracked = untracked
4193
+ advanced = True
4194
+
4195
+ # Re-stamp the spec's own baseline to the advanced one, on BOTH re-drive legs.
4196
+ #
4197
+ # The patch-restore leg needs it because the in-review route skips step-03 —
4198
+ # the only step that stamps `baseline_revision` — so without it the re-driven
4199
+ # step-04 would build its review diff (and, on an intent-gap/bad-spec
4200
+ # re-triage, revert) "since" the ORIGINAL pre-attempt sha, clawing back the
4201
+ # very resolve-session commits the advance above just blessed as the re-drive's
4202
+ # starting point.
4203
+ #
4204
+ # The from-scratch leg gets it too (#640a). Its step-03 re-stamps the key
4205
+ # itself, so the write is redundant on the happy path — but only ON that path:
4206
+ # until step-03 runs, the spec carries the escalated attempt's sha, and every
4207
+ # gate that reads a claimed baseline before then reads a stale one. The cost is
4208
+ # recorded rather than hidden: re-stamping removes the gate's INDEPENDENT
4209
+ # signal on this leg (it then compares a value the orchestrator itself wrote),
4210
+ # so a claim that genuinely diverged is journalled on the way out instead of
4211
+ # being silently normalized.
4212
+ #
4213
+ # Gated on `advanced`, not on truthiness of `task.baseline_commit`: a failed
4214
+ # advance leaves the OLD sha in that field, which passes a truthiness test
4215
+ # identically to a freshly advanced one. Writing it would make spec and task
4216
+ # agree on a stale value — the one state in which nothing downstream can tell
4217
+ # that the advance never happened, and the re-drive rebuilds from the wrong
4218
+ # point with no error anywhere. Skipping keeps the failure legible (the degrade
4219
+ # is journalled above) and keeps re-arm non-fatal outside a repo.
4220
+ #
4221
+ # Loud on WRITE failure: a silently stale spec baseline is exactly the hazard
4222
+ # being closed.
4223
+ #
4224
+ # Guarded on `is_file` FIRST, because a spec this process cannot reach is not a
4225
+ # write failure here — it is a SILENT one. Both frontmatter writers answer such a
4226
+ # path with `False` rather than an exception (`verify.set_frontmatter_status`,
4227
+ # `verify.set_frontmatter_field`), so without a check the re-stamp no-ops with
4228
+ # nothing on the record and the spec keeps the escalated attempt's sha.
4229
+ #
4230
+ # `task_spec_path` re-anchors the recorded path before we get here, which is what
4231
+ # makes `is_file` mean what it says. Resolved raw it meant something else and worse:
4232
+ # `spec_file` is persisted RELATIVE to the worktree for an isolated task, and the
4233
+ # main checkout carries the same layout, so the check passed on the wrong file and
4234
+ # the write landed there. The restore leg cannot reach any of this (its precondition
4235
+ # rejects a truthy `task.worktree_path`); the from-scratch leg has no such guard,
4236
+ # which is exactly why that precondition has to exist.
4237
+ #
4238
+ # `is_file` is necessary but not sufficient: a spec that EXISTS with no frontmatter
4239
+ # block also returns `False` from both writers. That shape is caught by the flip's
4240
+ # `flipped` check above and, here, by `overwritten` staying empty.
4241
+ if task.spec_file:
4242
+ spec_path = task_spec_path(task, state)
4243
+ if not spec_path.is_file():
4244
+ # OUTSIDE the `advanced` gate on purpose. Nesting this record inside it
4245
+ # made the two #640 legs shadow each other: on a project that is not a
4246
+ # repo the advance fails, `advanced` is False, and an unreadable spec
4247
+ # then produced NO record at all — the journal blamed git while the
4248
+ # status flip above had silently no-opped for an entirely different
4249
+ # reason. The two degrades compose; they do not substitute.
4250
+ journal.append(
4251
+ "rearm-baseline-restamp-skipped",
4252
+ story_key=key,
4253
+ spec_file=str(spec_path),
4254
+ baseline=task.baseline_commit or "",
4255
+ )
4256
+ elif advanced and task.baseline_commit:
4257
+ try:
4258
+ # Read through the same reader both consumers of a claimed baseline use,
4259
+ # so what gets journalled as "overwritten" is the value the gate would
4260
+ # have judged — not whichever key happened to be inspected here (#716).
4261
+ #
4262
+ # INSIDE the try, with the write it describes. `read_frontmatter` opens
4263
+ # the file itself, so an OSError here would otherwise escape as a
4264
+ # traceback from the one block whose whole contract is to turn a spec
4265
+ # this re-arm cannot move into an actionable `RearmError`. What it does
4266
+ # NOT rescue: `read_frontmatter` DEGRADES an unparseable YAML block to
4267
+ # `{}` rather than raising, so on such a spec `overwritten` is `""`, the
4268
+ # guard below is falsy, and no divergence record is written even though
4269
+ # the insert lands. That is the reader's deliberate observe-degrade
4270
+ # contract, not something to defeat here — the value is unknowable, and
4271
+ # inventing one would be worse than the silence.
4272
+ overwritten = auto_dev_baseline_of(verify.read_frontmatter(spec_path))
4273
+ verify.set_frontmatter_field(
4274
+ spec_path,
4275
+ "baseline_revision",
4276
+ task.baseline_commit,
4277
+ confine_root=task_spec_root(task, state),
4278
+ )
4279
+ except (OSError, UnicodeDecodeError, verify.FrontmatterWriteError) as e:
4280
+ # FrontmatterWriteError joins the tuple rather than getting its own
4281
+ # arm: the remedy is the same sentence ("fix the file"), and the
4282
+ # exception already says which shape it could not move. What matters
4283
+ # is that it aborts here — the stale-baseline hazard this block exists
4284
+ # to close is exactly what a swallowed write would leave behind.
4285
+ #
4286
+ # ...and that the abort leaves the spec as this re-arm FOUND it. This is
4287
+ # the LAST of the two refusals that can fire after a write has landed —
4288
+ # the flip and the result strip are both behind us, `save_state` is not —
4289
+ # so it carries the undo the sequenced refusals get for free (the other
4290
+ # is the spec block's `(OSError, UnicodeDecodeError)` arm, which the
4291
+ # strip raises through after the flip has published). Without
4292
+ # it a spec with a movable `status:` beside an unmovable
4293
+ # `baseline_revision:` came back flipped to the re-drive's status and
4294
+ # stripped of the terminal result, while the run still called the story
4295
+ # escalated.
4296
+ _restore_rearmed_spec(spec_path, spec_before, task, state)
4297
+ raise RearmError(
4298
+ f"cannot re-stamp baseline_revision on {spec_path} "
4299
+ f"({e.__class__.__name__}: {e}) — fix the file, then re-run resolve"
4300
+ ) from e
4301
+ if overwritten and overwritten != old_baseline:
4302
+ # Compared against `old_baseline` — what the RUN recorded for the
4303
+ # escalated attempt — NOT against `task.baseline_commit`, which the
4304
+ # advance above has already moved to the new HEAD. Measuring against the
4305
+ # advanced value made this fire on every ordinary from-scratch re-arm
4306
+ # whose resolve session committed anything: the spec and the run agreed
4307
+ # exactly, and the operator was still told they diverged. A record that
4308
+ # fires on the routine case is the "trains the operator to scroll past
4309
+ # the meaningful one" failure the `restore` split exists to prevent.
4310
+ #
4311
+ # What survives is the real signal, on BOTH legs: the spec claimed a
4312
+ # baseline the run never recorded. That is the only trace left of a
4313
+ # divergence the gate can no longer report, because the re-stamp is
4314
+ # about to normalize it away.
4315
+ journal.append(
4316
+ "rearm-baseline-restamped",
4317
+ story_key=key,
4318
+ spec_file=str(spec_path),
4319
+ overwritten=overwritten,
4320
+ baseline=task.baseline_commit,
4321
+ restore=bool(restore_patch),
4322
+ )
4323
+
4324
+ save_state(run_dir, state)
4325
+ journal.append(
4326
+ "story-escalation-resolved",
4327
+ story_key=key,
4328
+ baseline=task.baseline_commit or "",
4329
+ restore=bool(restore_patch),
4330
+ )
4331
+ return key
4332
+
4333
+
4334
+ def journal_entries_or_none(run_dir: Path) -> list[dict[str, Any]] | None:
4335
+ """This run's journal entries, or ``None`` when the journal cannot be read.
4336
+
4337
+ The re-arm surfaces read the journal TWICE to diff what a re-arm appended, and
4338
+ before that echo existed they read it not at all — so `Journal.entries()`' strict
4339
+ UTF-8 decode would turn a corrupt journal into a re-arm the operator can no longer
4340
+ perform, which is strictly worse than the missing echo and a regression against the
4341
+ gesture's own history. Shared by `cli.cmd_resolve` and `TuiApp._do_rearm` rather
4342
+ than living on one of them: the CLI's copy was left unguarded when the TUI's was
4343
+ hardened, and the CLI's echo now runs from a `finally`, where a raise would replace
4344
+ the `RearmError` the operator actually needs to see.
4345
+
4346
+ ``None`` rather than ``[]`` because the two callers DIFF two reads. Degrading a
4347
+ failed FIRST read to ``[]`` sets the watermark to zero, and a second read that
4348
+ succeeds then replays every historical `rearm-*`/`stale-restore-*` entry as if this
4349
+ re-arm had just produced it. A caller that cannot establish both ends of the diff
4350
+ must skip the echo, not guess at it.
4351
+ """
4352
+ try:
4353
+ # Non-mapping lines are dropped HERE so the annotation is true for every
4354
+ # caller: `Journal.entries()` appends `json.loads(line)` with no shape filter,
4355
+ # so a bare `3` or `null` on its own line survives as a non-dict entry and its
4356
+ # `list[dict[str, Any]]` return type is a claim about first-party producers,
4357
+ # not a guarantee — pyright sees `Any` and is satisfied. Both reads apply the
4358
+ # same filter, so the `len(before)` watermark stays exact.
4359
+ return [e for e in Journal(run_dir).entries() if isinstance(e, dict)]
4360
+ except (OSError, UnicodeDecodeError):
4361
+ return None
4362
+
4363
+
4364
+ def _journal_sequence(value: Any) -> tuple[Any, ...]:
4365
+ """A journal list field read back as a sequence, whatever the line actually held.
4366
+
4367
+ Every read in `rearm_event_notice` runs inside both operator surfaces' `finally`,
4368
+ where a `TypeError` replaces the outcome the operator needs — on the TUI, whose
4369
+ `_do_rearm` runs on Textual's message loop with no `_handle_exception` override,
4370
+ it ends the app. `", ".join` and `len` are the two reads that raise on a shape the
4371
+ journal admits (`"files": 3`, `"files": null`, `[1, 2]`); every sibling read is
4372
+ already `str()`-wrapped or f-string-interpolated and cannot.
4373
+
4374
+ A bare string is deliberately NOT iterated: `", ".join("abc")` renders `"a, b, c"`,
4375
+ which is worse than useless. It is wrapped as a single element instead, and `None`
4376
+ — which `.get(key, default)` returns whenever the key EXISTS holding null, so the
4377
+ default never applies — reads as empty.
4378
+ """
4379
+ if isinstance(value, (list, tuple)):
4380
+ return tuple(value)
4381
+ return () if value is None else (value,)
4382
+
4383
+
4384
+ def rearm_event_notice(
4385
+ entry: dict[str, Any],
4386
+ ) -> tuple[Literal["note", "warning"], str, str] | None:
4387
+ """`(severity, message, next_step)` for a re-arm record an operator must see.
4388
+
4389
+ ONE table, two surfaces. `cli._echo_rearm_events` prints `message` followed by
4390
+ `next_step`; `TuiApp._do_rearm` shows `message` alone. That split is the whole
4391
+ reason this returns three fields instead of a formatted line: the TUI re-arms and
4392
+ RESUMES in a single gesture, so an instruction to check something "before
4393
+ resuming" is already unactionable by the time it renders — but the finding it
4394
+ reports is not, and dropping the record to avoid the dead imperative is what left
4395
+ the TUI silent on three kinds `resolve` echoed.
4396
+
4397
+ Returns None for journal kinds no operator has to act on, so a caller can walk
4398
+ every new entry and let the table decide.
4399
+
4400
+ Severity is `"note"` or `"warning"`; each surface maps those onto its own channel.
4401
+ """
4402
+ if not isinstance(entry, dict):
4403
+ return None
4404
+ kind = entry.get("kind", "")
4405
+ if kind == "stale-restore-excluded":
4406
+ files = ", ".join(str(f) for f in _journal_sequence(entry.get("files")))
4407
+ return (
4408
+ "note",
4409
+ f"excluded the abandoned restore's new files from the re-drive baseline: {files}",
4410
+ "",
4411
+ )
4412
+ if kind == "stale-restore-unparseable":
4413
+ return (
4414
+ "warning",
4415
+ f"could not read the abandoned restore patch ({entry.get('patch', '?')}) "
4416
+ "— its new files may be swept into the next commit",
4417
+ "check `git status` before resuming",
4418
+ )
4419
+ if kind == "stale-restore-commits":
4420
+ n = len(_journal_sequence(entry.get("commits")))
4421
+ return (
4422
+ "warning",
4423
+ f"{n} commit(s) sit below the re-drive's new baseline "
4424
+ f"({str(entry.get('old_baseline', '?'))[:12]}..) — if any came from the "
4425
+ "abandoned attempt rather than your resolve, revert them now",
4426
+ "",
4427
+ )
4428
+ if kind == "rearm-baseline-advance-failed":
4429
+ return (
4430
+ "warning",
4431
+ f"could not advance the re-drive baseline ({entry.get('error', '?')}) — it "
4432
+ f"still names {str(entry.get('baseline', '') or '(none)')[:12]}, so the "
4433
+ "re-drive rebuilds against the tree as it stood before your resolve; the "
4434
+ "spec was deliberately NOT re-stamped",
4435
+ "Check the baseline before resuming",
4436
+ )
4437
+ if kind == "rearm-spec-write-unreachable":
4438
+ # ONE kind, TWO remedies, told apart by the `redrive` field its producer writes
4439
+ # — the live isolation mode of the re-drive, which this reader runs too late and
4440
+ # in the wrong process to determine for itself. A record predating the field is
4441
+ # an ISOLATED one: that was the only shape the producer could journal before the
4442
+ # in-place arm existed, so the absent field is a known value, not an unknown.
4443
+ spec = entry.get("spec_file", "?")
4444
+ if str(entry.get("redrive", "isolated") or "isolated") == "in-place":
4445
+ # The mirror shape: `isolation` was edited to `"none"` while the escalation
4446
+ # was paused, so the writes went into the mount the escalated attempt
4447
+ # recorded and the re-drive reads the main checkout instead. Committing is
4448
+ # not the remedy here and naming a branch would be actively wrong — the
4449
+ # in-place re-drive reads a WORKING TREE, so the edit simply has to be made
4450
+ # in the checkout the run resumes into.
4451
+ return (
4452
+ "warning",
4453
+ f"this run's isolation policy changed to `none` while the story was "
4454
+ f"escalated, so the re-arm's spec writes ({spec}) landed in the "
4455
+ "escalated attempt's worktree while the re-drive now runs in the main "
4456
+ "checkout — re-apply the correction to the main checkout's copy of the "
4457
+ "spec or the story re-wedges on the escalated attempt's status",
4458
+ "Correct the spec in the main checkout before resuming",
4459
+ )
4460
+ # The branch is the half an operator cannot infer: the re-drive cuts its fresh
4461
+ # worktree from the run's PINNED target branch, so a correction committed on
4462
+ # whatever the main checkout happens to have checked out is not the one it
4463
+ # reads. Named only when the record carries it — a run predating the field
4464
+ # leaves it empty, and a remedy that names no ref beats one that names a guess.
4465
+ base = str(entry.get("target_branch", "") or "")
4466
+ where = f" on `{base}`" if base else ""
4467
+ return (
4468
+ "warning",
4469
+ f"the re-drive of this story will mount a fresh worktree, so the re-arm's "
4470
+ f"spec writes ({spec}) land in a tree it discards — the re-driven session "
4471
+ "reads the COMMITTED spec, so commit the corrected "
4472
+ f"spec{where} or the story re-wedges on the escalated attempt's status",
4473
+ f"Commit the corrected spec{where} before resuming",
4474
+ )
4475
+ if kind == "rearm-upstream-write-unreachable":
4476
+ # The sentinel counterpart, and ONE remedy rather than the two above: the
4477
+ # producer only reaches this record on the mounting leg, because an in-place
4478
+ # re-drive reads the very checkout `resolve.run_session` ran the agent in. So
4479
+ # there is no `redrive` discriminator to read and no in-place arm to get wrong.
4480
+ #
4481
+ # It names the FOLDER, not a file, because the correction is not one file: the
4482
+ # skill sends the agent to `SPEC.md` or to this story's entry in `stories.yaml`,
4483
+ # and which of the two moved is the agent's choice, not something a journal
4484
+ # reader can recover. Naming both and the folder they sit in is what makes the
4485
+ # remedy actionable without claiming more than the record proves.
4486
+ root = str(entry.get("stories_root", "?"))
4487
+ base = str(entry.get("target_branch", "") or "")
4488
+ where = f" on `{base}`" if base else ""
4489
+ return (
4490
+ "warning",
4491
+ f"the sentinel was cleared, but the re-drive of this story will mount a "
4492
+ f"fresh worktree and re-plan from the COMMITTED tree — the upstream "
4493
+ f"correction in {root} (`SPEC.md` / `stories.yaml`) is uncommitted there, "
4494
+ f"so the re-plan reads the same intent that wedged and mints the sentinel "
4495
+ "again",
4496
+ f"Commit the corrected SPEC.md / stories.yaml{where} before resuming",
4497
+ )
4498
+ if kind == "rearm-spec-flip-skipped":
4499
+ # ONE kind, TWO outcomes, told apart by the flag the producer writes rather
4500
+ # than by anything readable from here: `rearm_escalation` raises `RearmError`
4501
+ # right after journalling this only when the flip failed on the very copy the
4502
+ # re-drive reads. It also journals it — and completes — when that copy is
4503
+ # unreadable from this process, or is a worktree-local file the re-drive
4504
+ # discards. This row used to claim the abort unconditionally, which told an
4505
+ # operator whose re-arm had SUCCEEDED that it "was REFUSED" and sent them to
4506
+ # add a `status:` to a file the re-drive never opens.
4507
+ spec = entry.get("spec_file", "?")
4508
+ status = entry.get("status", "?")
4509
+ if entry.get("refused"):
4510
+ # The message names the refusal rather than predicting a re-wedge, because
4511
+ # there is no re-drive left to wedge — and the next_step is the repair, not
4512
+ # an inspection, for the same reason.
4513
+ return (
4514
+ "warning",
4515
+ f"the recorded spec for this story ({spec}) could not be re-opened to "
4516
+ f"`{status}` — it carries no frontmatter `status:` to set, so the "
4517
+ "re-arm was REFUSED rather than re-driving a session that would wedge "
4518
+ "on the status it reads",
4519
+ "Add a top-level `status:` to the spec, then re-run resolve",
4520
+ )
4521
+ # No next_step, and deliberately: on this leg there is nothing to do to THIS
4522
+ # file. Whether anything is left to do at all is decided by the committed spec,
4523
+ # and `rearm-spec-write-unreachable` — journalled from the same block, on
4524
+ # exactly the legs where the committed spec is not already at the target —
4525
+ # carries that imperative, and holds the resume behind it.
4526
+ return (
4527
+ "warning",
4528
+ f"the recorded spec for this story ({spec}) could not be re-opened to "
4529
+ f"`{status}` — the re-arm was NOT refused, because that copy is not what "
4530
+ "the re-driven session reads: it mounts a fresh worktree and reads the "
4531
+ "COMMITTED spec",
4532
+ "",
4533
+ )
4534
+ if kind == "rearm-baseline-restamp-skipped":
4535
+ return (
4536
+ "warning",
4537
+ f"the recorded spec for this story ({entry.get('spec_file', '?')}) is not a "
4538
+ "readable file from here, so the baseline re-stamp was skipped — the spec "
4539
+ "still names the escalated attempt's baseline",
4540
+ "Check the recorded spec path before resuming",
4541
+ )
4542
+ if kind == "rearm-baseline-restamped":
4543
+ head = (
4544
+ f"re-stamped the spec baseline "
4545
+ f"{str(entry.get('overwritten', '?'))[:12]}.. -> "
4546
+ f"{str(entry.get('baseline', '?'))[:12]}.."
4547
+ )
4548
+ # NOT differentiated on the `restore` flag any more. That split predated the
4549
+ # record's condition moving to `overwritten != old_baseline` (compared against
4550
+ # what the RUN recorded, not against the just-advanced value): the record now
4551
+ # fires ONLY when the spec claimed a baseline the run never recorded, which is
4552
+ # equally exceptional on both legs. Keeping the split meant the patch-restore
4553
+ # leg's real divergence was the one downgraded to a note. The flag stays ON the
4554
+ # record because it says which leg produced it — not how routine it is.
4555
+ return (
4556
+ "warning",
4557
+ f"{head} — the spec claimed a DIFFERENT baseline than the run recorded, "
4558
+ "and this re-stamp is the only trace of it; the gate can no longer report "
4559
+ "that divergence",
4560
+ "",
4561
+ )
4562
+ return None
4563
+
4564
+
4565
+ def rearm_holds_the_resume(entry: dict[str, Any]) -> bool:
4566
+ """True for a re-arm record whose remedy has to land BEFORE the re-drive reads the
4567
+ tree — so a surface that re-arms and resumes in ONE gesture must stop after the
4568
+ re-arm and leave `froid-loop resume` to the operator.
4569
+
4570
+ TWO kinds qualify, and the discriminator is PROOF, not urgency.
4571
+ `rearm-spec-write-unreachable` is written only once `_redrive_spec_status` has
4572
+ established that the committed spec does NOT carry the status the re-drive routes
4573
+ on, and only for a spec the working-tree flip cannot reach. Resuming on it is not
4574
+ risky, it is futile: the re-drive discards the worktree, mounts a fresh one from
4575
+ git, and step-01 reads a status it cannot route — `unrecognized status in existing
4576
+ story file` halts it blocked, and the escalation is spent. The record's own
4577
+ next_step already said "commit the corrected spec before resuming"; both default
4578
+ surfaces then resumed in the same breath, which made the imperative unactionable at
4579
+ the moment it rendered. The interactive resolve agent cannot close that gap either
4580
+ — its skill forbids it from committing.
4581
+
4582
+ `rearm-upstream-write-unreachable` earns it the same way on the sentinel path,
4583
+ where there is no spec write to measure at all: the sentinel is cleared by
4584
+ deletion, and the correction that stops it recurring sits upstream in `SPEC.md` /
4585
+ `stories.yaml`. Its proof is `_redrive_reads_the_upstream_artifacts`, which fires
4586
+ the record only while the ref the re-drive mounts from does NOT already hold this
4587
+ checkout's copy of those two files — so, exactly as above, resuming is not risky
4588
+ but futile: the re-drive re-plans from a tree that never saw the correction and
4589
+ mints the same sentinel again.
4590
+
4591
+ The other warnings stay advisory and do NOT hold. `stale-restore-commits`,
4592
+ `stale-restore-unparseable` and `rearm-baseline-advance-failed` each report
4593
+ something an operator may need to act on, but none of them PROVES the re-drive
4594
+ cannot route, and holding on a maybe would turn the ordinary degrade path into a
4595
+ two-command gesture for an outcome nothing decided.
4596
+
4597
+ Not folded into `rearm_event_notice`'s tuple, because they are different questions
4598
+ asked of the same entry: that table answers "what do I tell the operator", this
4599
+ answers "may this gesture still resume". Both surfaces ask both, in one walk.
4600
+ """
4601
+ return isinstance(entry, dict) and entry.get("kind") in (
4602
+ "rearm-spec-write-unreachable",
4603
+ "rearm-upstream-write-unreachable",
4604
+ )
4605
+
4606
+
4607
+ def _stale_restore_residue(
4608
+ repo: Path,
4609
+ journal: Journal,
4610
+ story_key: str,
4611
+ old_latch: str | None,
4612
+ old_baseline: str | None,
4613
+ ) -> set[str]:
4614
+ """The untracked files an abandoned patch-restore attempt left in the tree —
4615
+ to be subtracted from the re-arm's refreshed `baseline_untracked` (issue #90).
4616
+
4617
+ Empty when no restore was latched. Deliberately *not* a `git apply -R`: the
4618
+ re-drive's own reset already reverts the patch's tracked hunks, an `apply -R`
4619
+ fails outright on any drift the resolve session introduced, and it misbehaves
4620
+ on the committed variant below. Only the patch's new files are durable
4621
+ contamination, and naming them is enough — `verify.safe_rollback` deletes
4622
+ whatever the refreshed snapshot stops blessing.
4623
+
4624
+ Also journals (warn-only) the commits sitting between the OLD baseline and the
4625
+ new one: a commit the escalated re-drive session made now becomes the next
4626
+ re-drive's permanent starting point, and no reset revisits it. It is not
4627
+ mechanically reversible — the resolve session's own blessed commits live in the
4628
+ same range and reverting those would claw back the human's resolution — so the
4629
+ human is the classifier. `froid-loop resolve` echoes these to stderr.
4630
+
4631
+ Best-effort throughout: a deleted or unreadable patch, a non-repo project, a
4632
+ bad old baseline — none may wedge a resolve. Every failure degrades to the
4633
+ pre-#90 behavior and says so in the journal.
4634
+ """
4635
+ if not old_latch:
4636
+ return set()
4637
+ patch_path = verify.resolve_restore_path(old_latch, repo)
4638
+
4639
+ residue: set[str] = set()
4640
+ try:
4641
+ residue = verify.patch_new_files(patch_path)
4642
+ except (OSError, UnicodeDecodeError) as e:
4643
+ # degrade to the pre-#90 snapshot rather than wedge the resolve
4644
+ journal.append(
4645
+ "stale-restore-unparseable",
4646
+ story_key=story_key,
4647
+ patch=str(patch_path),
4648
+ error=f"{e.__class__.__name__}: {e}",
4649
+ )
4650
+ else:
4651
+ if residue:
4652
+ journal.append(
4653
+ "stale-restore-excluded",
4654
+ story_key=story_key,
4655
+ patch=str(patch_path),
4656
+ files=sorted(residue),
4657
+ )
4658
+
4659
+ # Independent of the parse above — an unreadable patch must not also cost the
4660
+ # human the only notice they get about the committed variant.
4661
+ if old_baseline:
4662
+ try:
4663
+ shas = verify.commits_above(repo, old_baseline)
4664
+ except Exception: # nosec B110 - warn-only, must not fail re-arm
4665
+ shas = []
4666
+ if shas:
4667
+ journal.append(
4668
+ "stale-restore-commits",
4669
+ story_key=story_key,
4670
+ old_baseline=old_baseline,
4671
+ commits=shas,
4672
+ )
4673
+ return residue
4674
+
4675
+
4676
+ def _sentinel_condition(spec_path: Path, story_key: str) -> str | None:
4677
+ """The blocking condition (``unresolved`` / ``ambiguous``) iff ``spec_path`` is
4678
+ a fixed-slug pre-planning-halt sentinel for ``story_key``, else None."""
4679
+ from .stories import SENTINEL_SLUGS
4680
+
4681
+ for slug in SENTINEL_SLUGS:
4682
+ if spec_path.name == f"{story_key}-{slug}.md":
4683
+ return slug
4684
+ return None
4685
+
4686
+
4687
+ def _clear_sentinel(
4688
+ run_dir: Path, journal: Journal, spec_path: Path, story_key: str, sentinel_kind: str
4689
+ ) -> None:
4690
+ """Preserve a copy of the sentinel under ``{run_dir}/sentinels/`` (a write-only
4691
+ breadcrumb of what blocked planning), journal ``sentinel-cleared`` — carrying
4692
+ both the fixed slug (``sentinel_kind``) and the *recorded blocking condition*
4693
+ parsed from the sentinel's ``## Auto Run Result`` (the reason planning halted) —
4694
+ then delete the sentinel so the next dispatch is clean."""
4695
+ from .stories import recorded_blocking_condition
4696
+
4697
+ dest_dir = run_dir / "sentinels"
4698
+ dest_dir.mkdir(parents=True, exist_ok=True)
4699
+ condition = ""
4700
+ if spec_path.is_file():
4701
+ try:
4702
+ condition = recorded_blocking_condition(spec_path.read_text(encoding="utf-8"))
4703
+ except (OSError, UnicodeDecodeError):
4704
+ # An unreadable/binary sentinel still gets preserved+deleted so re-arm
4705
+ # completes; we just journal an empty blocking condition.
4706
+ condition = ""
4707
+ shutil.copy2(spec_path, dest_dir / spec_path.name)
4708
+ spec_path.unlink()
4709
+ journal.append(
4710
+ "sentinel-cleared",
4711
+ story_key=story_key,
4712
+ sentinel_kind=sentinel_kind,
4713
+ condition=condition,
4714
+ sentinel=spec_path.name,
4715
+ )