froid-loop 0.11.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- froid_loop/__init__.py +11 -0
- froid_loop/__main__.py +12 -0
- froid_loop/adapters/__init__.py +3 -0
- froid_loop/adapters/base.py +254 -0
- froid_loop/adapters/entrypoints.py +63 -0
- froid_loop/adapters/env_fault.py +290 -0
- froid_loop/adapters/generic.py +2013 -0
- froid_loop/adapters/mock.py +49 -0
- froid_loop/adapters/multiplexer.py +914 -0
- froid_loop/adapters/opencode_http.py +1687 -0
- froid_loop/adapters/profile.py +650 -0
- froid_loop/adapters/psmux_backend.py +1428 -0
- froid_loop/adapters/registry.py +322 -0
- froid_loop/adapters/tmux_backend.py +35 -0
- froid_loop/adapters/tmux_base.py +630 -0
- froid_loop/checks.py +187 -0
- froid_loop/cli.py +5041 -0
- froid_loop/data/__init__.py +0 -0
- froid_loop/data/froid_loop_hook.py +228 -0
- froid_loop/data/froid_loop_probe_hook.py +88 -0
- froid_loop/data/plugins/example/plugin.toml +21 -0
- froid_loop/data/plugins/tea/plugin.toml +184 -0
- froid_loop/data/plugins/tea/tea_plugin.py +258 -0
- froid_loop/data/plugins/unity/plugin.toml +140 -0
- froid_loop/data/plugins/unity/unity_assets/FroidLoop.Unity.Editor.asmdef +16 -0
- froid_loop/data/plugins/unity/unity_assets/FroidLoop.Unity.Editor.asmdef.meta +7 -0
- froid_loop/data/plugins/unity/unity_assets/SceneAutoSaveGuard.cs +221 -0
- froid_loop/data/plugins/unity/unity_assets/SceneAutoSaveGuard.cs.meta +11 -0
- froid_loop/data/plugins/unity/unity_assets/_folders/Editor.meta +8 -0
- froid_loop/data/plugins/unity/unity_assets/_folders/FroidLoop.meta +8 -0
- froid_loop/data/plugins/unity/unity_cleanup.py +125 -0
- froid_loop/data/plugins/unity/unity_dialog_probe.py +239 -0
- froid_loop/data/plugins/unity/unity_facts.md +17 -0
- froid_loop/data/plugins/unity/unity_plugin.py +415 -0
- froid_loop/data/plugins/unity/unity_quiesce.py +234 -0
- froid_loop/data/plugins/unity/unity_ready.py +230 -0
- froid_loop/data/plugins/unity/unity_seed_assets.py +298 -0
- froid_loop/data/plugins/unity/unity_setup.py +551 -0
- froid_loop/data/plugins/unity/unity_teardown.py +362 -0
- froid_loop/data/profiles/antigravity.toml +52 -0
- froid_loop/data/profiles/claude.toml +85 -0
- froid_loop/data/profiles/codex.toml +22 -0
- froid_loop/data/profiles/copilot.toml +52 -0
- froid_loop/data/profiles/gemini.toml +26 -0
- froid_loop/data/profiles/opencode.toml +54 -0
- froid_loop/data/settings/core.toml +458 -0
- froid_loop/data/skills/README.md +93 -0
- froid_loop/data/skills/froid-loop-resolve/SKILL.md +288 -0
- froid_loop/data/skills/froid-loop-setup/SKILL.md +161 -0
- froid_loop/data/skills/froid-loop-setup/assets/module-help.csv +3 -0
- froid_loop/data/skills/froid-loop-setup/assets/module.yaml +19 -0
- froid_loop/data/skills/froid-loop-sweep/SKILL.md +100 -0
- froid_loop/data/skills/froid-loop-sweep/automation-mode.md +127 -0
- froid_loop/data/skills/froid-loop-sweep/deferred-work-format.md +302 -0
- froid_loop/data/skills/froid-loop-sweep/migration-mode.md +86 -0
- froid_loop/decisions.py +202 -0
- froid_loop/deferredwork.py +2282 -0
- froid_loop/devcontract.py +892 -0
- froid_loop/diagnostics.py +1104 -0
- froid_loop/documents.py +532 -0
- froid_loop/engine.py +7732 -0
- froid_loop/envvars.py +111 -0
- froid_loop/escalation.py +225 -0
- froid_loop/events.py +266 -0
- froid_loop/fences.py +103 -0
- froid_loop/froidconfig.py +226 -0
- froid_loop/frontmatter.py +526 -0
- froid_loop/gates.py +133 -0
- froid_loop/install.py +2936 -0
- froid_loop/journal.py +178 -0
- froid_loop/machine.py +148 -0
- froid_loop/model.py +898 -0
- froid_loop/operatoractions.py +474 -0
- froid_loop/platform_util.py +1490 -0
- froid_loop/plugins/__init__.py +64 -0
- froid_loop/plugins/bus.py +259 -0
- froid_loop/plugins/context.py +319 -0
- froid_loop/plugins/loader.py +145 -0
- froid_loop/plugins/manifest.py +279 -0
- froid_loop/plugins/model.py +296 -0
- froid_loop/plugins/registry.py +245 -0
- froid_loop/plugins/trust.py +75 -0
- froid_loop/policy.py +1569 -0
- froid_loop/probe.py +1044 -0
- froid_loop/process_host.py +408 -0
- froid_loop/recovery_flow.py +1561 -0
- froid_loop/resolve.py +283 -0
- froid_loop/runs.py +4715 -0
- froid_loop/runsetup.py +1293 -0
- froid_loop/sanitize.py +593 -0
- froid_loop/settings_schema.py +276 -0
- froid_loop/signals.py +160 -0
- froid_loop/sprintstatus.py +609 -0
- froid_loop/statemachine.py +57 -0
- froid_loop/stories.py +615 -0
- froid_loop/stories_engine.py +796 -0
- froid_loop/sweep.py +1892 -0
- froid_loop/tokens.py +196 -0
- froid_loop/tui/__init__.py +11 -0
- froid_loop/tui/app.py +1584 -0
- froid_loop/tui/data.py +840 -0
- froid_loop/tui/launch.py +1003 -0
- froid_loop/tui/screens/__init__.py +1 -0
- froid_loop/tui/screens/dashboard.py +1071 -0
- froid_loop/tui/screens/modals.py +943 -0
- froid_loop/tui/screens/settings_screen.py +477 -0
- froid_loop/tui/settings.py +135 -0
- froid_loop/tui/widgets.py +981 -0
- froid_loop/verify.py +4545 -0
- froid_loop/workspace.py +320 -0
- froid_loop/worktree_flow.py +2301 -0
- froid_loop-0.11.1.dist-info/METADATA +728 -0
- froid_loop-0.11.1.dist-info/RECORD +116 -0
- froid_loop-0.11.1.dist-info/WHEEL +4 -0
- froid_loop-0.11.1.dist-info/entry_points.txt +2 -0
- froid_loop-0.11.1.dist-info/licenses/LICENSE +30 -0
froid_loop/runs.py
ADDED
|
@@ -0,0 +1,4715 @@
|
|
|
1
|
+
"""Run-directory discovery and helpers shared by the CLI and the TUI."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import contextlib
|
|
6
|
+
import contextvars
|
|
7
|
+
import hashlib
|
|
8
|
+
import json
|
|
9
|
+
import math
|
|
10
|
+
import os
|
|
11
|
+
import re
|
|
12
|
+
import secrets
|
|
13
|
+
import shutil
|
|
14
|
+
import stat
|
|
15
|
+
import sys
|
|
16
|
+
import tarfile
|
|
17
|
+
import time
|
|
18
|
+
from collections.abc import Iterable, Mapping
|
|
19
|
+
from dataclasses import dataclass
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
from typing import Any, Literal
|
|
22
|
+
|
|
23
|
+
from . import devcontract, envvars, verify
|
|
24
|
+
from .adapters.multiplexer import (
|
|
25
|
+
MultiplexerError,
|
|
26
|
+
TerminalMultiplexer,
|
|
27
|
+
get_multiplexer,
|
|
28
|
+
mux_usable,
|
|
29
|
+
)
|
|
30
|
+
from .frontmatter import auto_dev_baseline_of, parse_frontmatter, status_of
|
|
31
|
+
from .journal import STATE_FILE, VERIFY_DIR, Journal, load_state, save_state
|
|
32
|
+
from .model import PAUSE_ESCALATION, Phase, RunState, StoryTask
|
|
33
|
+
from .platform_util import (
|
|
34
|
+
MAX_SEGMENT,
|
|
35
|
+
UnconfinedWriteError,
|
|
36
|
+
_mkstemp_beside,
|
|
37
|
+
atomic_replace,
|
|
38
|
+
atomic_write_bytes_confined,
|
|
39
|
+
atomic_write_text_confined,
|
|
40
|
+
create_exclusive_confined,
|
|
41
|
+
has_parent_ref,
|
|
42
|
+
is_absolute_path,
|
|
43
|
+
is_link_like,
|
|
44
|
+
names_tree_root,
|
|
45
|
+
retrying_unlink,
|
|
46
|
+
safe_segment,
|
|
47
|
+
)
|
|
48
|
+
from .process_host import ProcessHostError, get_process_host
|
|
49
|
+
|
|
50
|
+
# The multiplexer registry's directory name inside a project's state subtree (see
|
|
51
|
+
# `mux_registry_root`). It sits beside the run entries and must never BE one: the
|
|
52
|
+
# leading underscore is what makes that structural, since `RUN_ID_RE` requires an
|
|
53
|
+
# alphanumeric first character, so no `--run-id` can key its state dir onto the
|
|
54
|
+
# registry. That is also what lets the orphan-state sweep tell the two apart by
|
|
55
|
+
# name alone (see `reconcile_orphan_state_dirs`).
|
|
56
|
+
MUX_REGISTRY_DIR = "_mux"
|
|
57
|
+
# psmux's own registry-root variable. Named here, in transport-agnostic code, for
|
|
58
|
+
# the same reason `PROJECT_OPTION` is: the export has to happen ahead of backend
|
|
59
|
+
# selection, which probes a subprocess, so it cannot be routed through a backend
|
|
60
|
+
# instance. See `export_psmux_registry_root`.
|
|
61
|
+
PSMUX_DATA_DIR = "PSMUX_DATA_DIR"
|
|
62
|
+
RUNS_DIR = Path(".froid-loop") / "runs"
|
|
63
|
+
ARCHIVE_DIR = Path(".froid-loop") / "archive"
|
|
64
|
+
PID_FILE = "engine.pid"
|
|
65
|
+
# Cross-process channel for a stop request: a control file the requester (CLI/TUI)
|
|
66
|
+
# writes and the engine reads. The body carries a `mode` — "graceful" or "hard".
|
|
67
|
+
#
|
|
68
|
+
# graceful (`stop --graceful`): finish the in-flight item, then finalize and stop.
|
|
69
|
+
# Honored at item boundaries only; resumable.
|
|
70
|
+
# hard (`stop`): stop now. Lodged by stop_run *before* it signals, honored by the
|
|
71
|
+
# engine at item boundaries and mid-session by the adapter wait loop.
|
|
72
|
+
#
|
|
73
|
+
# The file exists because signals are not a portable stop channel: there is no
|
|
74
|
+
# SIGUSR1 on Windows/psmux, and an inter-process SIGTERM is never delivered to a
|
|
75
|
+
# native-Windows engine at all, so the win32 "graceful" terminate is a no-op that
|
|
76
|
+
# only ever burned _STOP_WAIT_S into a force-kill (#319). SIGTERM remains the POSIX
|
|
77
|
+
# fast path — the file is what makes a stop work everywhere else. The engine stays
|
|
78
|
+
# the single writer of journal.jsonl, and the single *consumer* of this file;
|
|
79
|
+
# requesters only ever write it, adapters only ever read it.
|
|
80
|
+
STOP_REQUEST_FILE = "stop-request.json"
|
|
81
|
+
# The host-exec config baseline's name inside a run's state dir (see
|
|
82
|
+
# `config_digest_path_for`). A bare hex digest, not JSON: one opaque token, and a
|
|
83
|
+
# format an operator can read with `cat`.
|
|
84
|
+
CONFIG_DIGEST_FILE = "config-digest"
|
|
85
|
+
# Read cap for the file above. A sha256 hex digest is 64 bytes; the slack is for
|
|
86
|
+
# a trailing newline and for saying "this is not the digest" out of a file that
|
|
87
|
+
# is merely wrong rather than hostile. The cap's real job is the hostile case —
|
|
88
|
+
# see `read_trusted_config_digest` on why a bound, not a bigger buffer.
|
|
89
|
+
_MAX_DIGEST_BYTES = 256
|
|
90
|
+
_INVALID_PID_IDENTITY = -1.0 # impossible process start/create time; forces "not ours"
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
class StopRunError(Exception):
|
|
94
|
+
"""A live run could not be stopped — the engine honored neither channel (the
|
|
95
|
+
lodged stop request nor SIGTERM) and its pid's identity can no longer be
|
|
96
|
+
verified, so force-killing would risk an unrelated (reused) pid. The caller
|
|
97
|
+
surfaces this rather than silently marking stopped."""
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
class GracefulStopError(Exception):
|
|
101
|
+
"""A graceful-stop request could not be lodged (run already finished, or its
|
|
102
|
+
engine is provably dead so the request would never be consumed). ``str()`` is
|
|
103
|
+
the operator-facing message the CLI/TUI surface verbatim."""
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
class LiveSessionError(Exception):
|
|
107
|
+
"""A run directory was not removed because the run's agent session is still
|
|
108
|
+
live (see :func:`live_session_may_be_ours`). ``str()`` is the operator-facing
|
|
109
|
+
message the CLI/TUI surface verbatim."""
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
# How long stop_run waits for a signalled engine to exit before falling back to
|
|
113
|
+
# marking the run stopped itself.
|
|
114
|
+
_STOP_WAIT_S = 10.0
|
|
115
|
+
_STOP_POLL_S = 0.1
|
|
116
|
+
# How long stop_run lets a force-kill settle before deciding it failed. A kill that
|
|
117
|
+
# returns cleanly is not proof of death — win32 shells `taskkill /F /T` with
|
|
118
|
+
# `check=False`, so a refused kill raises nothing — but the pid can also linger for a
|
|
119
|
+
# moment after a delivered SIGKILL, and `is_alive` is a bare existence probe that
|
|
120
|
+
# reads a not-yet-reaped process as alive. Long enough to outlast that, short enough
|
|
121
|
+
# that a genuinely surviving engine is still noticed while the operator waits.
|
|
122
|
+
_KILL_CONFIRM_S = 0.5
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def new_run_id() -> str:
|
|
126
|
+
return time.strftime("%Y%m%d-%H%M%S") + "-" + secrets.token_hex(2)
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
# A run id is a lookup key with exactly one legitimate producer (new_run_id), and it
|
|
130
|
+
# lands in three positions at once: a directory name under RUNS_DIR, a multiplexer
|
|
131
|
+
# session name (froid-loop-<id>), and a git ref component (froid-loop/<id>/<unit>).
|
|
132
|
+
# So an id supplied from outside is *rejected*, never sanitized — coercing it would
|
|
133
|
+
# break the id<->path<->session bijection the CLI relies on to find a run again.
|
|
134
|
+
#
|
|
135
|
+
# The charset is a superset of every new_run_id() output and excludes, by
|
|
136
|
+
# construction: path separators and `..` (traversal), `<>:"|?*` plus trailing dots
|
|
137
|
+
# and spaces (Windows), `.` and `:` (multiplexer session-name mangling), and all
|
|
138
|
+
# whitespace/control characters. It is also identity under safe_ref_segment, so the
|
|
139
|
+
# unit branch a run produces reads back verbatim — hence no ref check below.
|
|
140
|
+
RUN_ID_RE = re.compile(r"[A-Za-z0-9][A-Za-z0-9_-]*")
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def is_valid_run_id(value: str) -> bool:
|
|
144
|
+
"""True when ``value`` is a run id we would have produced ourselves — the guard
|
|
145
|
+
every externally-supplied ``--run-id`` and every id recomposed from the outside
|
|
146
|
+
world (a foreign multiplexer session name) must pass before it touches a path.
|
|
147
|
+
|
|
148
|
+
The length cap is ``platform_util.MAX_SEGMENT``: a run id is a directory name.
|
|
149
|
+
The ``safe_segment`` identity check adds the one rule ``RUN_ID_RE`` cannot
|
|
150
|
+
express — the reserved Windows device basenames (``CON``, ``NUL``, ``COM1``…),
|
|
151
|
+
which are legal-looking ids that no filesystem will accept as a directory.
|
|
152
|
+
|
|
153
|
+
The control-session shape (``ctl``, ``ctl-…``, any letter case) is reserved
|
|
154
|
+
on the same principle, against the multiplexer namespace instead of the
|
|
155
|
+
filesystem's — see :func:`is_reserved_run_id` for the shape and why case is
|
|
156
|
+
folded. Refusing the id here is what makes the two session namespaces
|
|
157
|
+
disjoint: every agent session is ``froid-loop-<valid id>``, so none can
|
|
158
|
+
reach the control session's name."""
|
|
159
|
+
return _wellformed_run_id(value) and not is_reserved_run_id(value)
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def _wellformed_run_id(value: str) -> bool:
|
|
163
|
+
"""The shape half of :func:`is_valid_run_id`: charset, length, and the
|
|
164
|
+
reserved-device-basename identity check — everything except the
|
|
165
|
+
control-session reservation. Split out because the *parse* side
|
|
166
|
+
(:func:`_agent_run_id`) must accept ids the *mint* refuses: a run
|
|
167
|
+
persisted by an older release under e.g. ``ctl-foo`` owns a genuine
|
|
168
|
+
``froid-loop-ctl-foo`` agent session that the sweep has to be able to
|
|
169
|
+
reach."""
|
|
170
|
+
return (
|
|
171
|
+
bool(RUN_ID_RE.fullmatch(value))
|
|
172
|
+
and len(value) <= MAX_SEGMENT
|
|
173
|
+
and safe_segment(value) == value
|
|
174
|
+
)
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def is_reserved_run_id(value: str) -> bool:
|
|
178
|
+
"""The MINT-side reservation: any id of the control-session shape (``ctl``
|
|
179
|
+
or ``ctl-…``, any letter case) is refused at :func:`is_valid_run_id`.
|
|
180
|
+
Deliberately broader than :func:`run_id_aliases_control_session` — a new id
|
|
181
|
+
anywhere near the control namespace buys nothing but confusion, so none is
|
|
182
|
+
admitted — while the read paths, which must handle ids an older release
|
|
183
|
+
already persisted, use the narrow test. ``RUN_ID_RE`` is ASCII-only, so
|
|
184
|
+
``str.lower`` is the exact fold (see the narrow test for why case folds at
|
|
185
|
+
all)."""
|
|
186
|
+
v = value.lower()
|
|
187
|
+
return v == "ctl" or v.startswith("ctl-")
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def run_id_aliases_control_session(value: str) -> bool:
|
|
191
|
+
"""True when ``session_name(value)`` names a session that can BE a live
|
|
192
|
+
control session: the fixed name (id ``ctl``) or a per-registry digest name
|
|
193
|
+
(id ``ctl-<16 hex>`` — the only suffix :func:`ctl_session_for` can mint).
|
|
194
|
+
The adapter's ensure-session would *adopt* that live session as the run's
|
|
195
|
+
own, and the run's teardown would kill the whole control session, every
|
|
196
|
+
parked window of every run in it — so the project-free READ paths key on
|
|
197
|
+
this: :func:`kill_session` skips such an id, :func:`_agent_run_id`
|
|
198
|
+
refuses to read such a session as a run, and ``cli``/the TUI refuse to
|
|
199
|
+
resume/re-arm/replan such a run. This is the SHAPE question — "could
|
|
200
|
+
this name be a control session's on some registry" — and it must stay
|
|
201
|
+
out of any site asking the *instance* question ("is it the control
|
|
202
|
+
session this process addresses"): :func:`live_session_may_be_ours`
|
|
203
|
+
compares against the actual names (the fixed one plus this project's
|
|
204
|
+
:func:`ctl_session_for`), because discounting the whole shape there
|
|
205
|
+
destroyed run dirs under live `ctl-<other digest>` agents on tmux.
|
|
206
|
+
|
|
207
|
+
Compared **case-insensitively**: psmux resolves a session by opening
|
|
208
|
+
``<data dir>\\<name>.port`` by name (``src/paths.rs:113``, source-read at
|
|
209
|
+
v3.3.8), and NTFS opens names case-insensitively — measured: with
|
|
210
|
+
``froid-loop-ctl-x`` live, target ``froid-loop-CTL-x`` answers
|
|
211
|
+
``has-session``, is refused as a duplicate by ``new-session``, and a kill
|
|
212
|
+
through it takes the lowercase session down.
|
|
213
|
+
|
|
214
|
+
Deliberately narrower than :func:`is_reserved_run_id`: a historical
|
|
215
|
+
``ctl-foo`` run's session is a GENUINE agent session, distinct from every
|
|
216
|
+
control session and addressable exactly and safely (tmux: measured, the
|
|
217
|
+
exact full target removes only it; our seam sends ``=``-exact targets —
|
|
218
|
+
``tmux_base.py:141,166``, source-read. psmux: exact port files, case
|
|
219
|
+
aside). Skipping those too made such runs unreachable by ``stop`` and
|
|
220
|
+
``cleanup`` both. Ceiling, named: an id of exactly the digest shape whose
|
|
221
|
+
hex is NOT the current registry's digest is also skipped — undecidable
|
|
222
|
+
without the project in hand, and the leak direction (one stale session
|
|
223
|
+
left standing) is the safe one."""
|
|
224
|
+
return is_ctl_session_name(session_name(value).lower())
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def is_parsable_run_id(value: str) -> bool:
|
|
228
|
+
"""The PARSE-side counterpart of :func:`is_valid_run_id`: may an id
|
|
229
|
+
recovered from an existing multiplexer name be acted on as a run?
|
|
230
|
+
|
|
231
|
+
The two questions are different and must never share a predicate.
|
|
232
|
+
:func:`is_valid_run_id` answers "may a NEW id be this", so it carries the
|
|
233
|
+
mint's broad ctl reservation (:func:`is_reserved_run_id`) — and a reader
|
|
234
|
+
that borrows it stops recognising every id an older release already
|
|
235
|
+
persisted. A ``ctl-foo`` run minted before that reservation owns a real
|
|
236
|
+
run dir and a real ``run-ctl-foo`` control-session window; asking the
|
|
237
|
+
mint's question about them leaks both, unreachable by the sweep forever.
|
|
238
|
+
|
|
239
|
+
So: the shape half (:func:`_wellformed_run_id` — charset, length, and the
|
|
240
|
+
reserved-device-basename check, because the id still steers a run-dir
|
|
241
|
+
path) minus only the narrow alias test
|
|
242
|
+
(:func:`run_id_aliases_control_session`), which the read paths key on
|
|
243
|
+
because reading one of THOSE as a run points a kill path at the control
|
|
244
|
+
plane. Exactly :func:`_agent_run_id`'s guard, public so the other parse
|
|
245
|
+
sites ask it instead of re-deriving it — the ctl-window sweep in
|
|
246
|
+
``tui.launch`` did borrow the mint's, and parked pre-upgrade windows
|
|
247
|
+
leaked from ``cleanup`` because of it."""
|
|
248
|
+
return _wellformed_run_id(value) and not run_id_aliases_control_session(value)
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def list_run_dirs(project: Path) -> list[Path]:
|
|
252
|
+
"""All run dirs containing a state.json, oldest first (run ids sort
|
|
253
|
+
chronologically)."""
|
|
254
|
+
runs = project / RUNS_DIR
|
|
255
|
+
if not runs.is_dir():
|
|
256
|
+
return []
|
|
257
|
+
return sorted(d for d in runs.iterdir() if (d / "state.json").is_file())
|
|
258
|
+
|
|
259
|
+
|
|
260
|
+
def all_run_dirs(project: Path) -> list[Path] | None:
|
|
261
|
+
"""Every run dir under the runs root — ``state.json`` or not — oldest first,
|
|
262
|
+
or ``None`` when the listing could not be taken.
|
|
263
|
+
|
|
264
|
+
The ungated counterpart to :func:`list_run_dirs`, and the one to ask when the
|
|
265
|
+
question is "does a run still own its control plane" rather than "which runs
|
|
266
|
+
can I read". A run whose state.json was removed or corrupted still holds a
|
|
267
|
+
live ``engine.pid``, so the gated view walks straight past exactly the run an
|
|
268
|
+
operator is mid-recovery on — the hazard :func:`_run_dir_names` documents,
|
|
269
|
+
whose set this wraps rather than re-listing.
|
|
270
|
+
|
|
271
|
+
``None`` is an unreadable runs root and means *nothing was learned*, which is
|
|
272
|
+
not the same answer as the empty list a missing root gives. Callers that act
|
|
273
|
+
on "no live runs" have to tell those apart; see :func:`_run_dir_names`.
|
|
274
|
+
"""
|
|
275
|
+
names = _run_dir_names(project)
|
|
276
|
+
if names is None:
|
|
277
|
+
return None
|
|
278
|
+
root = project / RUNS_DIR
|
|
279
|
+
return sorted(root / name for name in names)
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
def latest_run_dir(project: Path) -> Path | None:
|
|
283
|
+
candidates = list_run_dirs(project)
|
|
284
|
+
return candidates[-1] if candidates else None
|
|
285
|
+
|
|
286
|
+
|
|
287
|
+
def write_named_pid(pidfile: Path, pid: int) -> None:
|
|
288
|
+
"""Record ``pid`` plus its identity to ``pidfile``, so a later liveness read can
|
|
289
|
+
tell our process from a stranger that inherited a reused pid (immediate on
|
|
290
|
+
Windows). One whitespace-delimited line: ``"<pid>"`` (legacy) or
|
|
291
|
+
``"<pid> <identity>"``; the identity token is omitted when the platform can't
|
|
292
|
+
provide one. The parameterized form :func:`write_pid` builds on — reused for the
|
|
293
|
+
Unity dialog probe's own ``unity-dialog-probe.pid`` handle."""
|
|
294
|
+
identity = get_process_host().identity(pid)
|
|
295
|
+
line = f"{pid} {identity}" if identity is not None else str(pid)
|
|
296
|
+
pidfile.write_text(line, encoding="utf-8")
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
def write_pid(run_dir: Path) -> None:
|
|
300
|
+
"""Record the engine pid plus its identity, so a later liveness read can tell
|
|
301
|
+
our engine from a stranger that inherited a reused pid (immediate on Windows).
|
|
302
|
+
Never deleted: a stale pid that reads as gone is the signal a run was
|
|
303
|
+
interrupted."""
|
|
304
|
+
write_named_pid(run_dir / PID_FILE, os.getpid())
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def session_name(run_id: str) -> str:
|
|
308
|
+
return f"froid-loop-{run_id}"
|
|
309
|
+
|
|
310
|
+
|
|
311
|
+
def attach_target_argv(target: str) -> list[str]:
|
|
312
|
+
"""Multiplexer command to reach a target session/window (see
|
|
313
|
+
:meth:`TerminalMultiplexer.attach_target_argv`)."""
|
|
314
|
+
return get_multiplexer().attach_target_argv(target)
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
def session_target(run_id: str) -> str:
|
|
318
|
+
"""Seam-canonical target token for the run's agent session (see
|
|
319
|
+
:meth:`TerminalMultiplexer.target`)."""
|
|
320
|
+
return get_multiplexer().target(session_name(run_id))
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
def attach_argv(run_id: str) -> list[str]:
|
|
324
|
+
return attach_target_argv(session_target(run_id))
|
|
325
|
+
|
|
326
|
+
|
|
327
|
+
# ------------------------------------------------------- user-scoped state root
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
class StateRootError(Exception):
|
|
331
|
+
"""No user-scoped state root could be derived from this environment — every
|
|
332
|
+
candidate base was unset, empty, relative, or named the filesystem root. The
|
|
333
|
+
control plane has nowhere to live, and the caller must fail rather than guess
|
|
334
|
+
(see :func:`state_root`)."""
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
def _state_base(value: str | None) -> Path | None:
|
|
338
|
+
"""``value`` as a usable base directory, or ``None`` when it cannot be one.
|
|
339
|
+
|
|
340
|
+
The single rule every *derived* candidate below is held to, so the POSIX and
|
|
341
|
+
win32 branches cannot drift into judging their inputs differently. A base is
|
|
342
|
+
rejected when it is unset, empty, relative, or names the filesystem root
|
|
343
|
+
itself. The last three are the answers a broken environment gives *instead* of
|
|
344
|
+
raising, which is what makes them worth naming:
|
|
345
|
+
|
|
346
|
+
- **empty**: ``os.path.expanduser("~")`` answers ``""`` on Windows for a
|
|
347
|
+
set-but-empty ``USERPROFILE``, and ``Path("")`` is the current directory.
|
|
348
|
+
- **relative**: including ``"~"`` itself, which is what ``expanduser`` returns
|
|
349
|
+
when it cannot expand at all. The state root would then move with the
|
|
350
|
+
launch cwd, and a run whose control plane it cannot find again is a run
|
|
351
|
+
that stalls to ``session_timeout_min`` rather than one that fails.
|
|
352
|
+
- **the root**: ``expanduser("~")`` answers ``"/"`` on POSIX for a set-but-empty
|
|
353
|
+
``HOME`` (``posixpath`` folds the empty prefix to the root), which would put
|
|
354
|
+
``/.local/state/froid-loop`` on the filesystem root — a permission error for
|
|
355
|
+
an ordinary user and, for a containerised root, a silent write to ``/``.
|
|
356
|
+
``base == base.parent`` is the root test on both flavours.
|
|
357
|
+
|
|
358
|
+
``os.path.isabs`` rather than :func:`platform_util.is_absolute_path`: the
|
|
359
|
+
latter is purpose-built for "must stay inside the project" guards and is
|
|
360
|
+
strictly broader — it calls the drive-*relative* ``C:foo`` absolute, which is
|
|
361
|
+
exactly the value that must not become a state root. The question here is the
|
|
362
|
+
platform's own, and each branch below only ever runs on its own platform.
|
|
363
|
+
"""
|
|
364
|
+
if not value or not os.path.isabs(value):
|
|
365
|
+
return None
|
|
366
|
+
base = Path(value)
|
|
367
|
+
return None if base == base.parent else base
|
|
368
|
+
|
|
369
|
+
|
|
370
|
+
def state_root() -> Path:
|
|
371
|
+
"""The froid-loop state root for this user: the out-of-tree home of per-run
|
|
372
|
+
control-plane state — the events channel (#494) and, later, the config digest
|
|
373
|
+
(#498). Outside the project tree because a branch switch, a worktree mount or
|
|
374
|
+
a rollback must not be able to take a live run's control plane away.
|
|
375
|
+
|
|
376
|
+
Resolution, first answer wins:
|
|
377
|
+
|
|
378
|
+
1. ``FROID_LOOP_STATE_DIR``, used as the state root **itself** — no
|
|
379
|
+
``froid-loop`` segment is appended, because the variable names our root
|
|
380
|
+
rather than a base to build one under. It is honoured as spelled (see
|
|
381
|
+
:func:`envvars.state_dir`) and is not passed through ``_state_base``:
|
|
382
|
+
*skipping* a stated override would be a silent countermand, where skipping
|
|
383
|
+
a derived base only moves on to the next guess.
|
|
384
|
+
|
|
385
|
+
It must still be **absolute**, and a relative spelling raises rather than
|
|
386
|
+
being resolved for the operator. Absoluteness is not a matter of taste
|
|
387
|
+
here — the root is read by two processes with different working
|
|
388
|
+
directories. The engine exports it to the session as
|
|
389
|
+
``FROID_LOOP_EVENTS_DIR`` and the multiplexer launches that session at
|
|
390
|
+
``spec.cwd`` (a worktree under isolation), while the watcher polls it from
|
|
391
|
+
the orchestrator's own cwd. A relative root therefore names two different
|
|
392
|
+
directories at once: the relay writes its Stop where nothing is watching,
|
|
393
|
+
and the run waits out ``session_timeout_min`` — the exact silent stall
|
|
394
|
+
``_state_base`` rejects relative *derived* bases to avoid, and the one
|
|
395
|
+
this whole channel was moved out of the tree to prevent.
|
|
396
|
+
|
|
397
|
+
Raising is not the countermand the paragraph above refuses: it names the
|
|
398
|
+
variable and the fix, where absolutizing against whichever cwd this
|
|
399
|
+
process happens to have would be the guess. The not-the-root half of
|
|
400
|
+
``_state_base``'s rule is deliberately *not* applied — that half exists to
|
|
401
|
+
stop a broken environment's ``""`` from landing a guess at ``/``, and an
|
|
402
|
+
override is not a guess.
|
|
403
|
+
2. POSIX — ``$XDG_STATE_HOME/froid-loop`` when that variable names an absolute
|
|
404
|
+
path, else ``~/.local/state/froid-loop``. A relative ``XDG_STATE_HOME`` is
|
|
405
|
+
*ignored*, which the XDG base-directory spec requires of its consumers.
|
|
406
|
+
(``install._shield_inherited_excludes`` resolves a relative
|
|
407
|
+
``XDG_CONFIG_HOME`` instead of ignoring it — the opposite call for the
|
|
408
|
+
opposite reason: there we reproduce *git's* reading of the variable, here
|
|
409
|
+
we are the spec's own consumer.)
|
|
410
|
+
3. win32 — ``%LOCALAPPDATA%\\froid-loop\\state``, else
|
|
411
|
+
``%USERPROFILE%\\AppData\\Local\\froid-loop\\state``. ``LOCALAPPDATA`` names
|
|
412
|
+
the per-user, per-machine, non-roaming store Windows intends for exactly
|
|
413
|
+
this, and the second form is its documented default location.
|
|
414
|
+
|
|
415
|
+
**Never** ``Path.home()`` on the win32 arm. It is ``ntpath.expanduser("~")``,
|
|
416
|
+
which prefers ``USERPROFILE`` and then falls back to ``HOMEDRIVE`` +
|
|
417
|
+
``HOMEPATH`` — a pair that on a domain-joined machine may name a network home
|
|
418
|
+
share. A control plane whose atomic renames and ``O_NOFOLLOW``-anchored writes
|
|
419
|
+
live on an SMB share is not the local directory this needs, and the derivation
|
|
420
|
+
also disagrees with the one git uses for its own ``$HOME``
|
|
421
|
+
(``install._shield_home_git_ignore`` documents that split in full). Reading
|
|
422
|
+
``LOCALAPPDATA``/``USERPROFILE`` directly asks for the store by name instead of
|
|
423
|
+
inferring it from a home.
|
|
424
|
+
|
|
425
|
+
Raises :class:`StateRootError` when no candidate answers. This is a write
|
|
426
|
+
path, so it raises rather than degrading to a plausible-looking default:
|
|
427
|
+
``platform_util.resolve_or_lexical`` states the doctrine (observation may
|
|
428
|
+
degrade, repair writes must raise), and the degraded outcomes here are all
|
|
429
|
+
silent — a control plane at the cwd, or at ``/``, that the *next* process to
|
|
430
|
+
ask resolves somewhere else.
|
|
431
|
+
"""
|
|
432
|
+
override = envvars.state_dir()
|
|
433
|
+
if override:
|
|
434
|
+
# `os.path.isabs` on the raw string, matching `_state_base` exactly rather
|
|
435
|
+
# than `Path.is_absolute` — the rule and its reason are stated there.
|
|
436
|
+
if not os.path.isabs(override):
|
|
437
|
+
raise StateRootError(
|
|
438
|
+
f"{envvars.STATE_DIR} must name an absolute directory: {override!r} is "
|
|
439
|
+
"relative, and the state root is read by both this process and the "
|
|
440
|
+
"session it launches — which run from different working directories, "
|
|
441
|
+
"so a relative root names two different places and the run's "
|
|
442
|
+
"completion signal is written where nothing is watching"
|
|
443
|
+
)
|
|
444
|
+
return Path(override)
|
|
445
|
+
if sys.platform == "win32":
|
|
446
|
+
local = _state_base(os.environ.get("LOCALAPPDATA"))
|
|
447
|
+
if local:
|
|
448
|
+
return local / "froid-loop" / "state"
|
|
449
|
+
profile = _state_base(os.environ.get("USERPROFILE"))
|
|
450
|
+
if profile:
|
|
451
|
+
return profile / "AppData" / "Local" / "froid-loop" / "state"
|
|
452
|
+
else:
|
|
453
|
+
xdg = _state_base(os.environ.get("XDG_STATE_HOME"))
|
|
454
|
+
if xdg:
|
|
455
|
+
return xdg / "froid-loop"
|
|
456
|
+
home = _state_base(os.path.expanduser("~"))
|
|
457
|
+
if home:
|
|
458
|
+
return home / ".local" / "state" / "froid-loop"
|
|
459
|
+
raise StateRootError(
|
|
460
|
+
"cannot locate a state directory for froid-loop's run control plane: "
|
|
461
|
+
+ (
|
|
462
|
+
"neither %LOCALAPPDATA% nor %USERPROFILE% names an absolute directory"
|
|
463
|
+
if sys.platform == "win32"
|
|
464
|
+
else "neither $XDG_STATE_HOME nor $HOME names an absolute directory"
|
|
465
|
+
)
|
|
466
|
+
+ f" — set {envvars.STATE_DIR} to the directory it should live in"
|
|
467
|
+
)
|
|
468
|
+
|
|
469
|
+
|
|
470
|
+
def project_state_root(project: Path) -> Path:
|
|
471
|
+
"""The subtree of :func:`state_root` holding every run of this project:
|
|
472
|
+
``<state root>/<project key>``. Split out from :func:`state_dir_for` because
|
|
473
|
+
the GC reads it as a *directory to enumerate* rather than composing one run's
|
|
474
|
+
path — see :func:`reconcile_orphan_state_dirs`, whose whole job is the entries
|
|
475
|
+
under here that no longer have a run dir."""
|
|
476
|
+
return state_root() / project_tag(project)
|
|
477
|
+
|
|
478
|
+
|
|
479
|
+
def mux_registry_root(project: Path) -> Path:
|
|
480
|
+
"""This project's terminal-multiplexer registry root:
|
|
481
|
+
``<state root>/<project key>/_mux`` (see :data:`MUX_REGISTRY_DIR`).
|
|
482
|
+
|
|
483
|
+
A *registry* is the directory a multiplexer keeps its per-session addressing
|
|
484
|
+
state in — psmux writes one ``.port``/``.key``/``.sid``/``.pid`` quartet per
|
|
485
|
+
session under ``PSMUX_DATA_DIR`` (default ``%USERPROFILE%\\.psmux``), and
|
|
486
|
+
every verb resolves a session by reading that quartet back. Two processes
|
|
487
|
+
that disagree about the root therefore disagree about which sessions exist,
|
|
488
|
+
which is why the root is *derived* — from the project, through the same
|
|
489
|
+
:func:`project_tag` every ownership tag already uses — rather than minted per
|
|
490
|
+
run, read from a file, or taken from whatever the launching shell exported.
|
|
491
|
+
See :func:`export_psmux_registry_root` for the export and its rules.
|
|
492
|
+
|
|
493
|
+
Keyed on the project rather than on froid-loop as a whole so a prune bug in
|
|
494
|
+
one project cannot address another project's servers at all: the partition
|
|
495
|
+
becomes structural instead of a filter (the ``@froid_project`` tag stays, as
|
|
496
|
+
the tmux-side answer and the belt). The price is that one ``psmux ls`` no
|
|
497
|
+
longer shows every froid-loop session on the machine — stated for the operator
|
|
498
|
+
in ``docs/multiplexer-backends.md`` and printed by ``froid-loop mux``.
|
|
499
|
+
|
|
500
|
+
Under :func:`state_root` and not in the project tree, deliberately: a branch
|
|
501
|
+
switch or a rollback that deleted a ``.port`` file would leave the server
|
|
502
|
+
alive, unreachable, and invisible to ``psmux ls`` in *any* registry — a
|
|
503
|
+
manufactured orphan. Same doctrine :func:`state_root` itself exists for.
|
|
504
|
+
"""
|
|
505
|
+
return project_state_root(project) / MUX_REGISTRY_DIR
|
|
506
|
+
|
|
507
|
+
|
|
508
|
+
def export_psmux_registry_root(project: Path) -> str | None:
|
|
509
|
+
"""Point this process — and everything it spawns — at ``project``'s registry
|
|
510
|
+
by exporting ``PSMUX_DATA_DIR``. Returns the value in force afterwards, or
|
|
511
|
+
``None`` when no root could be derived.
|
|
512
|
+
|
|
513
|
+
**The process environment, not a per-call argument.** The seam spawns every
|
|
514
|
+
psmux verb through ``BaseTmuxBackend._run``, whose ``env=None`` default means
|
|
515
|
+
*inherit this process's environment*, and a create-call-only injection is
|
|
516
|
+
worse than none: the session's server would come up under a root every later
|
|
517
|
+
``has_session`` / ``list_window_ids`` cannot see, and those verbs report an
|
|
518
|
+
unreadable registry as ``False`` / ``[]`` — a live run reading itself as gone.
|
|
519
|
+
One export ahead of dispatch covers every verb in-process.
|
|
520
|
+
|
|
521
|
+
**The root is always derived, and an ambient value never changes it.** That
|
|
522
|
+
is the whole rule, and the absence of an exception is the point:
|
|
523
|
+
:func:`mux_registry_root` is a pure function of (project, state root), so any
|
|
524
|
+
two froid-loop processes given the same project and the same state root agree
|
|
525
|
+
— which is the entire property #537 exists to establish. A value already in
|
|
526
|
+
the environment is *overridden*, and the caller says so
|
|
527
|
+
(:func:`cli._configure_mux` reports it once on stderr; ``froid-loop mux``
|
|
528
|
+
discloses it).
|
|
529
|
+
|
|
530
|
+
**Why an operator's own ``PSMUX_DATA_DIR`` is not honoured**, since honouring
|
|
531
|
+
it is the obvious kindness and it was tried:
|
|
532
|
+
|
|
533
|
+
- It would make the registry a function of the launch *shell*. A TUI started
|
|
534
|
+
from the Start menu carries no profile environment and derives; a run
|
|
535
|
+
started from a dev shell whose profile exports a root honours that root.
|
|
536
|
+
Two registries on one machine, and a live session reading as gone in one of
|
|
537
|
+
them — which is the failure this module exists to prevent, not a corner of
|
|
538
|
+
it.
|
|
539
|
+
- Whether honouring is even the right answer is unknowable from here. A
|
|
540
|
+
process that finds a root in its environment cannot tell one the operator
|
|
541
|
+
typed once in *this* shell — where a clean sibling process would derive —
|
|
542
|
+
from one their profile exports into *every* shell, where a clean sibling
|
|
543
|
+
honours it. The two produce byte-identical environments and want opposite
|
|
544
|
+
answers, so no comparison settles it: the missing fact is the operator's
|
|
545
|
+
intent, and it is not in the environment.
|
|
546
|
+
- It contradicts the promise made beside it. ``FROID_LOOP_STATE_DIR``'s
|
|
547
|
+
documentation says there is deliberately no second variable naming the
|
|
548
|
+
registry, because "two knobs that can disagree would put two processes on
|
|
549
|
+
different registries, each blind to the other's live sessions". An ambient
|
|
550
|
+
``PSMUX_DATA_DIR`` is exactly that second knob.
|
|
551
|
+
|
|
552
|
+
Overridden rather than *refused*, deliberately: ``PSMUX_DATA_DIR`` is psmux's
|
|
553
|
+
variable, and an operator may have it set for their own sessions with no
|
|
554
|
+
thought of froid-loop at all. Erroring out of every command on such a machine
|
|
555
|
+
would be froid-loop claiming a name it does not own. The remedy runs the other
|
|
556
|
+
way and ``froid-loop mux`` prints it ready to paste: point *your* shell at
|
|
557
|
+
froid-loop's root, which is a function of the project rather than of whichever
|
|
558
|
+
shell happened to launch something.
|
|
559
|
+
|
|
560
|
+
**Overridden, but not abandoned.** A machine that had an absolute value
|
|
561
|
+
exported before the upgrade kept its froid-loop sessions in THAT registry,
|
|
562
|
+
because the old backend simply inherited it — so the displaced root is
|
|
563
|
+
handed to :func:`~.adapters.psmux_backend.note_displaced_registry` here,
|
|
564
|
+
the last moment anything can still read it, and the migration sweep runs a
|
|
565
|
+
tag-scoped pass over it alongside psmux's default
|
|
566
|
+
(:meth:`~.adapters.psmux_backend.PsmuxMultiplexer.legacy_registries`).
|
|
567
|
+
Without that the override would strand exactly the sessions it displaced,
|
|
568
|
+
with cleanup reporting a clean machine.
|
|
569
|
+
|
|
570
|
+
Wanting one registry to serve both is a real request and is deliberately not
|
|
571
|
+
answered here. It needs a stated operator preference rather than a guess at
|
|
572
|
+
one — and it must be a policy *whether*, never a *where*: ``policy.toml`` is
|
|
573
|
+
written by the sessions this orchestrator drives, so a policy-sourced root
|
|
574
|
+
would let a driven session choose which registry the cleanup path kills in.
|
|
575
|
+
|
|
576
|
+
**No ``FROID_LOOP_*`` knob for the root either.** It is derived state, not
|
|
577
|
+
configuration; ``FROID_LOOP_STATE_DIR`` already relocates it transitively —
|
|
578
|
+
one knob, one cascade, instead of two that can disagree. And ``envvars.py``
|
|
579
|
+
gains no entry for ``PSMUX_DATA_DIR`` itself: that module is scoped to
|
|
580
|
+
``FROID_LOOP_*`` names and this is psmux's own, unregistered on the same
|
|
581
|
+
precedent as ``PSMUX_ALLOW_NESTING``.
|
|
582
|
+
|
|
583
|
+
**No root travels between processes.** Because every froid-loop process
|
|
584
|
+
derives its own root, nothing about a registry has to be transported at
|
|
585
|
+
all. What does have to travel is the *state root*: coding-CLI windows are
|
|
586
|
+
told it explicitly through their env dict (:func:`pinned_state_env`), and
|
|
587
|
+
everything else — a session's window-0 shell, the TUI's parked engine
|
|
588
|
+
windows — inherits it, as it always has. psmux's ``PSMUX_BARE_ENV=1`` mode
|
|
589
|
+
breaks that inheritance and is **not supported**: the psmux backend warns
|
|
590
|
+
once per process when it is on (see ``PsmuxMultiplexer._warn_if_bare_env``).
|
|
591
|
+
|
|
592
|
+
Never raises. This runs ahead of *every* command, ``diagnose`` and
|
|
593
|
+
``validate`` included, and an underivable state root must not take the
|
|
594
|
+
diagnostics down with it. ``None`` means "no root established": psmux keeps
|
|
595
|
+
whatever it had, which is also the root cleanup sweeps as the legacy one.
|
|
596
|
+
"""
|
|
597
|
+
try:
|
|
598
|
+
root = str(mux_registry_root(project))
|
|
599
|
+
except (StateRootError, OSError, RuntimeError):
|
|
600
|
+
# OSError/RuntimeError: project_tag resolves the project, which raises on
|
|
601
|
+
# a path the OS cannot canonicalize and, below 3.13, on a symlink loop.
|
|
602
|
+
# The ambient value is left exactly as found — there is nothing better to
|
|
603
|
+
# put there, and PsmuxMultiplexer._run still refuses to spawn under a
|
|
604
|
+
# value psmux would panic on.
|
|
605
|
+
return None
|
|
606
|
+
displaced = os.environ.get(PSMUX_DATA_DIR)
|
|
607
|
+
os.environ[PSMUX_DATA_DIR] = root
|
|
608
|
+
if displaced is not None and displaced != root:
|
|
609
|
+
# The variable is now gone, and it was the only record of where a
|
|
610
|
+
# pre-upgrade machine's sessions live: before #537 the backend simply
|
|
611
|
+
# inherited it. Hand it to the backend that has to sweep there, at the
|
|
612
|
+
# one moment it is still knowable. Imported here rather than at module
|
|
613
|
+
# scope because this is the psmux leaf, and this module talks to the
|
|
614
|
+
# seam — the coupling is confined to the function already named for
|
|
615
|
+
# psmux's own variable.
|
|
616
|
+
from .adapters.psmux_backend import note_displaced_registry
|
|
617
|
+
|
|
618
|
+
note_displaced_registry(displaced)
|
|
619
|
+
return root
|
|
620
|
+
|
|
621
|
+
|
|
622
|
+
def pinned_state_env() -> dict[str, str]:
|
|
623
|
+
"""``{FROID_LOOP_STATE_DIR: <this process's state root>}``, for a child that
|
|
624
|
+
must land on the same one — or ``{}`` when no root can be derived.
|
|
625
|
+
|
|
626
|
+
A convenience spelling of :func:`pin_state_root` over an empty dict, for
|
|
627
|
+
composing env dicts (the engine's session env spreads it in). The final
|
|
628
|
+
merge before a window launch goes through :func:`pin_state_root` itself —
|
|
629
|
+
a spread of this dict is only an ordering guarantee, and ordering
|
|
630
|
+
guarantees nothing when the dict is ``{}``.
|
|
631
|
+
|
|
632
|
+
**Resolved, never forwarded.** Passing this only when the operator set it
|
|
633
|
+
would leave exactly the default case broken, which is the common one. What
|
|
634
|
+
travels is the answer this process reached, however it reached it.
|
|
635
|
+
|
|
636
|
+
What follows the state root, and what does not, since the two are easy to
|
|
637
|
+
swap: the run's control plane (:func:`state_dir_for`), its event channel
|
|
638
|
+
(:func:`events_dir_for`) and the multiplexer registry
|
|
639
|
+
(:func:`mux_registry_root`) all live under it, so a child computing a
|
|
640
|
+
different root writes and reads where nothing else looks. The run *directory*
|
|
641
|
+
does not — :func:`run_dir_for` is in-tree at ``<project>/.froid-loop/runs``
|
|
642
|
+
and moves with the project, not with this.
|
|
643
|
+
|
|
644
|
+
``{}`` rather than a raise: a child told nothing derives its own answer and
|
|
645
|
+
fails on the same broken environment with its own message, which is better
|
|
646
|
+
than a launcher that cannot report anything at all.
|
|
647
|
+
"""
|
|
648
|
+
return pin_state_root({})
|
|
649
|
+
|
|
650
|
+
|
|
651
|
+
def pin_state_root(env: Mapping[str, str]) -> dict[str, str]:
|
|
652
|
+
"""``env`` with its ``FROID_LOOP_STATE_DIR`` entry forced to this process's
|
|
653
|
+
own answer: **set** to the resolved state root when one derives, **removed**
|
|
654
|
+
when none does. Other keys pass through untouched.
|
|
655
|
+
|
|
656
|
+
The chokepoint for every merge where a caller-supplied env (a profile's
|
|
657
|
+
``[env]`` table rides those dicts) meets the state-root pin — the engine's
|
|
658
|
+
coding-CLI window, the probe window, and the attached resolve session. A
|
|
659
|
+
"pin spreads last" ordering rule is not enough, because with an underivable
|
|
660
|
+
state root there is no pin key to order: :func:`pinned_state_env` is ``{}``
|
|
661
|
+
and a profile-declared absolute root would sail through, aiming the window
|
|
662
|
+
at a state root — and so a per-project registry — its own parent cannot
|
|
663
|
+
see. Removing the key instead makes the child inherit the parent's own
|
|
664
|
+
(broken) value and fail exactly as the parent fails: whatever a child
|
|
665
|
+
concludes is what a clean process under the same conditions concludes, in
|
|
666
|
+
the error arm too. The strip governs only what froid-loop *adds* to a
|
|
667
|
+
child; a value already in the environment a child inherits is not
|
|
668
|
+
scrubbed here.
|
|
669
|
+
"""
|
|
670
|
+
pinned = dict(env)
|
|
671
|
+
try:
|
|
672
|
+
pinned[envvars.STATE_DIR] = str(state_root())
|
|
673
|
+
except StateRootError:
|
|
674
|
+
pinned.pop(envvars.STATE_DIR, None)
|
|
675
|
+
return pinned
|
|
676
|
+
|
|
677
|
+
|
|
678
|
+
def state_dir_for(project: Path, run_id: str) -> Path:
|
|
679
|
+
"""This run's control-plane directory: ``<state root>/<project key>/<run id>``.
|
|
680
|
+
|
|
681
|
+
The project key is :func:`project_tag`, reused verbatim rather than re-derived:
|
|
682
|
+
it already resolves the project before digesting it, so the two spellings of
|
|
683
|
+
one project a caller can arrive with — a symlinked path, a relative one — key
|
|
684
|
+
to the same directory. They must, or a run started through one spelling would
|
|
685
|
+
write its events where a poll through the other never looks, and the run would
|
|
686
|
+
wait out ``session_timeout_min`` with the completion signal sitting on disk.
|
|
687
|
+
Its ``resolve()`` raising on a project the OS cannot canonicalize is correct
|
|
688
|
+
here for the same reason: an unknowable location cannot be keyed at all, and
|
|
689
|
+
guessing one is the wrong-directory write the tag exists to prevent.
|
|
690
|
+
|
|
691
|
+
``run_id`` needs no sanitizing — the id contract (see :data:`RUN_ID_RE`) is
|
|
692
|
+
already "a legal path segment on every platform", pinned by
|
|
693
|
+
:func:`is_valid_run_id`, and an id from outside is rejected there rather than
|
|
694
|
+
coerced here.
|
|
695
|
+
"""
|
|
696
|
+
return project_state_root(project) / run_id
|
|
697
|
+
|
|
698
|
+
|
|
699
|
+
def events_dir_for(project: Path, run_id: str) -> Path:
|
|
700
|
+
"""The run's hook-event channel: the directory the relay writes a session's
|
|
701
|
+
events into and ``SignalWatcher`` polls for them."""
|
|
702
|
+
return state_dir_for(project, run_id) / "events"
|
|
703
|
+
|
|
704
|
+
|
|
705
|
+
def config_digest_path_for(project: Path, run_id: str) -> Path:
|
|
706
|
+
"""The run's host-exec config baseline: ``runsetup.config_digest`` as of the
|
|
707
|
+
last time a human started or resumed this run (#498).
|
|
708
|
+
|
|
709
|
+
Out here rather than in ``state.json`` because the baseline exists to police
|
|
710
|
+
the agent-writable tree, and until this move it *lived* in it: a session that
|
|
711
|
+
rewrote ``policy.toml`` could blank or re-stamp the field in the same breath
|
|
712
|
+
and the warning `resume` owes the operator never fired. The same reasoning the
|
|
713
|
+
events channel moved on (#494).
|
|
714
|
+
|
|
715
|
+
**What moving it buys, stated exactly.** It closes the *incidental* path: the
|
|
716
|
+
pin is no longer a project file, so nothing a session does in the ordinary
|
|
717
|
+
course of rewriting the tree can collaterally blank it — which is the case the
|
|
718
|
+
advisory was documented to catch. It is **not** a boundary against a
|
|
719
|
+
deliberate one. Sessions run with permission bypass by default — every shipped
|
|
720
|
+
profile's ``bypass_args``, which ``GenericAdapter.interactive_argv`` uses
|
|
721
|
+
unless ``[adapter] extra_args`` overrides them; that is what an unattended loop
|
|
722
|
+
is — and are handed ``FROID_LOOP_EVENTS_DIR``, whose parent is this directory. A
|
|
723
|
+
session that goes looking can *truncate* this file and the reader below answers
|
|
724
|
+
``""`` — a real "no baseline" — or delete it and blank the in-tree copy
|
|
725
|
+
(``RunState.trusted_config_digest``, the secondary this falls back to) for the
|
|
726
|
+
same silence. Either way the result is indistinguishable from a run that never
|
|
727
|
+
had a baseline: any marker saying "this run *should* have one" would have to
|
|
728
|
+
live somewhere the same session cannot reach, and no such place exists at equal
|
|
729
|
+
privilege. Closing it needs privilege separation on the state root, not a better
|
|
730
|
+
hiding place — tracked in #571."""
|
|
731
|
+
return state_dir_for(project, run_id) / CONFIG_DIGEST_FILE
|
|
732
|
+
|
|
733
|
+
|
|
734
|
+
def read_trusted_config_digest(project: Path, run_id: str) -> str | None:
|
|
735
|
+
"""This run's persisted host-exec baseline, or ``None`` when the state root
|
|
736
|
+
holds none for it.
|
|
737
|
+
|
|
738
|
+
``None`` is "ask the in-tree copy", not "no pin" — the two are different
|
|
739
|
+
answers and the caller acts on the difference (see
|
|
740
|
+
``cli._resume_paused_run``). No file here means this run's baseline is
|
|
741
|
+
reachable only through ``state.json``: it was paused before #498, or the
|
|
742
|
+
project moved and keyed its state subtree somewhere new
|
|
743
|
+
(:func:`project_state_root`). An *empty* file, by contrast, is a real answer
|
|
744
|
+
of "no baseline" and comes back as ``""``.
|
|
745
|
+
|
|
746
|
+
**Known limit: a file at this key can be stale (#572).** The key is the
|
|
747
|
+
project's resolved path, so a project that moves away and later returns finds
|
|
748
|
+
its old subtree still here — nothing can sweep it in between (FEATURES.md) —
|
|
749
|
+
holding the baseline blessed before it left, while the blessing it picked up
|
|
750
|
+
in between is the one in ``state.json``. Preferring the file means that older
|
|
751
|
+
pin wins for one resume, which re-stamps this key and heals it. Preferring the
|
|
752
|
+
fresher-looking in-tree copy is *not* the fix: it is session-writable, so it
|
|
753
|
+
would hand any session the silencing #498 closed. Arbitrating by sequence
|
|
754
|
+
number needs a counterpart the session cannot forge, and at equal privilege
|
|
755
|
+
there is none — the same wall as #571, reached by re-keying instead of
|
|
756
|
+
tampering.
|
|
757
|
+
|
|
758
|
+
Pure observation, so it degrades rather than raising: a state root this host
|
|
759
|
+
cannot name, or a file it cannot read, both answer ``None`` and hand the
|
|
760
|
+
decision to the in-tree copy. The write half raises — see
|
|
761
|
+
:func:`write_trusted_config_digest` — and the split is the standard one
|
|
762
|
+
(``platform_util.resolve_or_lexical`` states the doctrine). Degrading here
|
|
763
|
+
costs at most one advisory warning; a resume that *aborts* because an
|
|
764
|
+
advisory could not be read would be the worse failure, and the resume is
|
|
765
|
+
about to resolve the same state root for its events channel anyway, where
|
|
766
|
+
the error is owned and reported.
|
|
767
|
+
|
|
768
|
+
**Deliberately not ``read_text``**, and for the same reason the write is
|
|
769
|
+
``follow_symlinks=False``: this file sits in a directory the driven session
|
|
770
|
+
can reach (its parent is the ``FROID_LOOP_EVENTS_DIR`` the engine exports), so
|
|
771
|
+
the *shape* of what is at the path has to be established before any bytes are
|
|
772
|
+
consumed. Degrading on a hostile path is not enough when the read itself is
|
|
773
|
+
the weapon:
|
|
774
|
+
|
|
775
|
+
* ``O_NONBLOCK`` + an ``S_ISREG`` check **on the descriptor**. Opening a FIFO
|
|
776
|
+
for reading otherwise blocks until someone writes — indefinitely — and
|
|
777
|
+
``resume`` is a foreground command a human is waiting on, so a planted FIFO
|
|
778
|
+
wedges the terminal rather than costing a warning. The check is on the fd,
|
|
779
|
+
not the path, so it cannot be raced: ``fstat`` describes the object actually
|
|
780
|
+
opened.
|
|
781
|
+
* ``O_NOFOLLOW``, so the name is read rather than wherever it points.
|
|
782
|
+
* At most :data:`_MAX_DIGEST_BYTES`. A link to an endless source
|
|
783
|
+
(``/dev/zero``) reads forever otherwise, and raises ``MemoryError`` — not
|
|
784
|
+
the ``OSError`` this promises never to leak. The cap removes the condition
|
|
785
|
+
instead of absorbing it.
|
|
786
|
+
|
|
787
|
+
The POSIX-only flags degrade to 0 on win32, which has neither FIFOs at these
|
|
788
|
+
paths nor ``O_NOFOLLOW``; the size cap and the regular-file check carry there
|
|
789
|
+
on their own. This mirrors ``tui.launch._read_ctl_window`` deliberately — same
|
|
790
|
+
hazard, same shape, one idiom. It does **not** collapse empty to ``None`` the
|
|
791
|
+
way that twin does: here the two are different answers (above).
|
|
792
|
+
|
|
793
|
+
None of this makes the baseline tamper-*proof* — a session can still delete
|
|
794
|
+
the file, and #571 carries that. It stops a tampered path from hanging or
|
|
795
|
+
exhausting the orchestrator, which is a different and fixable harm."""
|
|
796
|
+
try:
|
|
797
|
+
path = config_digest_path_for(project, run_id)
|
|
798
|
+
except (StateRootError, OSError, RuntimeError):
|
|
799
|
+
return None
|
|
800
|
+
flags = os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0) | getattr(os, "O_NONBLOCK", 0)
|
|
801
|
+
flags |= getattr(os, "O_BINARY", 0) # win32: no CRLF translation on the raw fd
|
|
802
|
+
try:
|
|
803
|
+
fd = os.open(path, flags)
|
|
804
|
+
except OSError:
|
|
805
|
+
return None
|
|
806
|
+
try:
|
|
807
|
+
if not stat.S_ISREG(os.fstat(fd).st_mode):
|
|
808
|
+
return None
|
|
809
|
+
data = os.read(fd, _MAX_DIGEST_BYTES)
|
|
810
|
+
except OSError:
|
|
811
|
+
return None
|
|
812
|
+
finally:
|
|
813
|
+
os.close(fd)
|
|
814
|
+
try:
|
|
815
|
+
return data.decode("utf-8").strip()
|
|
816
|
+
except UnicodeDecodeError:
|
|
817
|
+
return None
|
|
818
|
+
|
|
819
|
+
|
|
820
|
+
def write_trusted_config_digest(project: Path, run_id: str, digest: str) -> None:
|
|
821
|
+
"""Stamp ``digest`` as this run's host-exec baseline, creating the state dir.
|
|
822
|
+
|
|
823
|
+
Raises rather than degrading — a repair write, and a silently skipped stamp
|
|
824
|
+
is the outcome hardest to detect later: the next resume reads no file and
|
|
825
|
+
decides on the in-tree copy alone, which is the tree this baseline exists to
|
|
826
|
+
police. The caller is starting or resuming a run and is about to resolve the
|
|
827
|
+
very same state root for its events channel, so a root that cannot be named
|
|
828
|
+
or written fails that run regardless; failing here just fails it sooner,
|
|
829
|
+
before the pid lands.
|
|
830
|
+
|
|
831
|
+
**Call this only after the run dir exists.** Creating the state dir is what
|
|
832
|
+
makes this the earliest writer into it, and :func:`reconcile_orphan_state_dirs`
|
|
833
|
+
reads its entries *before* the live run-dir names on the strength of run dirs
|
|
834
|
+
being created strictly first — a state dir minted ahead of its run dir would
|
|
835
|
+
look like an orphan to a ``clean`` racing the launch."""
|
|
836
|
+
path = config_digest_path_for(project, run_id)
|
|
837
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
838
|
+
# Confined to the state root (#593): a machine-minted record under a root
|
|
839
|
+
# whose path the driven session is handed (FROID_LOOP_EVENTS_DIR names its
|
|
840
|
+
# sibling), so a planted link here must be replaced, never written through to
|
|
841
|
+
# whatever it aims at. Refusing a link at the FINAL component was not enough —
|
|
842
|
+
# `mkstemp(dir=...)` and `os.replace`'s destination still resolved every
|
|
843
|
+
# directory above by name, and the `mkdir` on the line above ACCEPTS a
|
|
844
|
+
# symlinked directory, so a link planted at either session-reachable component
|
|
845
|
+
# (`<project tag>/`, `<run id>/`) survived the setup step and redirected both
|
|
846
|
+
# the temp and the published stamp. `state_root()` is the one component the
|
|
847
|
+
# anchored walk starts from rather than checks, and it is a host fact this
|
|
848
|
+
# process derives — not a path any session names. The trailing newline is for
|
|
849
|
+
# the operator who cats the file.
|
|
850
|
+
atomic_write_text_confined(path, digest + "\n", confine_root=state_root())
|
|
851
|
+
|
|
852
|
+
|
|
853
|
+
# ---------------------------------------------------- run resolution / liveness
|
|
854
|
+
|
|
855
|
+
|
|
856
|
+
def run_dir_for(project: Path, run_id: str) -> Path:
|
|
857
|
+
return project / RUNS_DIR / run_id
|
|
858
|
+
|
|
859
|
+
|
|
860
|
+
def is_run(run_dir: Path) -> bool:
|
|
861
|
+
"""A directory is a run iff it holds a state.json."""
|
|
862
|
+
return (run_dir / STATE_FILE).is_file()
|
|
863
|
+
|
|
864
|
+
|
|
865
|
+
class RunRefError(Exception):
|
|
866
|
+
"""A run ref matched no run, or was ambiguous."""
|
|
867
|
+
|
|
868
|
+
|
|
869
|
+
def short_ref(run_id: str) -> str:
|
|
870
|
+
"""The trailing hex segment — the minimal handle users type."""
|
|
871
|
+
return run_id.rsplit("-", 1)[-1]
|
|
872
|
+
|
|
873
|
+
|
|
874
|
+
def _is_path_escape(ref: str) -> bool:
|
|
875
|
+
"""True when ``ref`` would steer ``run_dir_for``'s recomposition outside the
|
|
876
|
+
runs dir — it is absolute/drive-qualified, climbs with ``..``, names the runs
|
|
877
|
+
dir itself rather than anything inside it, or carries a path separator of
|
|
878
|
+
either flavour. Sub-check of the run-id charset rather than `is_valid_run_id`
|
|
879
|
+
itself: a run dir created by an older version (or by hand) may bear a name we
|
|
880
|
+
would no longer mint, and must stay addressable.
|
|
881
|
+
|
|
882
|
+
`names_tree_root` restores this site to the three-guard pairing every sibling
|
|
883
|
+
already spells (`policy.py`, `adapters/profile.py`, `plugins/manifest.py`); it
|
|
884
|
+
was the only member of the family omitting it (#480). It closes the spellings
|
|
885
|
+
that recompose to the runs *root* instead of a run in it. ``""`` and ``"."``
|
|
886
|
+
join to it exactly — measured here, both `runs / ""` and `runs / "."` *are*
|
|
887
|
+
the runs dir — so a `state.json` lying at that root made the exact branch
|
|
888
|
+
below hand `delete_run` the whole runs tree to `rmtree`. ``"..."``, ``".. "``
|
|
889
|
+
and ``" "`` are the Win32 half of the same rule (cited, not measurable on
|
|
890
|
+
POSIX): the trim of trailing periods and spaces leaves ``..`` or nothing, so
|
|
891
|
+
they name `.froid-loop/` or the runs dir there while both pure pathlib flavours
|
|
892
|
+
keep them as ordinary one-segment names.
|
|
893
|
+
|
|
894
|
+
Addressability is unharmed: skipping the exact branch only defers to partial
|
|
895
|
+
matching, and a legacy dir named ``"..."`` is still enumerated by
|
|
896
|
+
`list_run_dirs` and still matched by its own spelling.
|
|
897
|
+
|
|
898
|
+
`names_win32_alias`, the family's fourth member, is deliberately NOT applied
|
|
899
|
+
here — it would make a legacy run dir named ``NUL`` or ``run. `` permanently
|
|
900
|
+
unaddressable, which is the one thing this guard exists to prevent. Refusing
|
|
901
|
+
to *mint* such a name is `is_valid_run_id`'s job, and it already does it with
|
|
902
|
+
a `safe_segment` identity check."""
|
|
903
|
+
return (
|
|
904
|
+
is_absolute_path(ref)
|
|
905
|
+
or has_parent_ref(ref)
|
|
906
|
+
or names_tree_root(ref)
|
|
907
|
+
or "/" in ref
|
|
908
|
+
or "\\" in ref
|
|
909
|
+
)
|
|
910
|
+
|
|
911
|
+
|
|
912
|
+
def resolve_run_dir(project: Path, ref: str) -> Path:
|
|
913
|
+
"""Full or partial run id -> its run dir. An exact id wins outright;
|
|
914
|
+
otherwise a partial matches when the trailing segment starts with `ref` or
|
|
915
|
+
the full id ends with `ref` (run ids are date-prefixed, so the tail is what
|
|
916
|
+
distinguishes them). Raises RunRefError on no match / ambiguity.
|
|
917
|
+
|
|
918
|
+
The exact branch recomposes a path from the raw ref, so it is skipped for any
|
|
919
|
+
ref that could escape the runs dir (`froid-loop delete ../../x` would otherwise
|
|
920
|
+
rmtree an outside directory that happens to hold a state.json). Such a ref
|
|
921
|
+
falls through to partial matching, which can only ever yield a name
|
|
922
|
+
`list_run_dirs` enumerated — and so cannot escape.
|
|
923
|
+
|
|
924
|
+
An EMPTY ref is refused outright rather than deferred: `""` is a prefix and a
|
|
925
|
+
suffix of every name, so partial matching reads it as a wildcard — harmlessly
|
|
926
|
+
ambiguous with two runs, but silently resolving the sole run of a one-run
|
|
927
|
+
project, which handed `froid-loop delete ""` that run. No addressability is
|
|
928
|
+
lost (no directory can be named `""`); every other escape spelling keeps the
|
|
929
|
+
partial fallback so a legacy dir named `"..."` stays matchable by its own
|
|
930
|
+
spelling."""
|
|
931
|
+
if not ref:
|
|
932
|
+
raise RunRefError("empty run ref: it would match every run, never name one")
|
|
933
|
+
if not _is_path_escape(ref):
|
|
934
|
+
exact = run_dir_for(project, ref)
|
|
935
|
+
if is_run(exact):
|
|
936
|
+
return exact
|
|
937
|
+
matches = [
|
|
938
|
+
d
|
|
939
|
+
for d in list_run_dirs(project)
|
|
940
|
+
if short_ref(d.name).startswith(ref) or d.name.endswith(ref)
|
|
941
|
+
]
|
|
942
|
+
if not matches:
|
|
943
|
+
raise RunRefError(f"no such run: {ref}")
|
|
944
|
+
if len(matches) > 1:
|
|
945
|
+
listing = "\n".join(f" {d.name}" for d in matches)
|
|
946
|
+
raise RunRefError(f"ambiguous run ref {ref!r} matches {len(matches)} runs:\n{listing}")
|
|
947
|
+
return matches[0]
|
|
948
|
+
|
|
949
|
+
|
|
950
|
+
def read_pid(run_dir: Path) -> int | None:
|
|
951
|
+
"""The recorded engine pid, or None when missing/unparseable. Reads the first
|
|
952
|
+
whitespace token, tolerating both the legacy pid-only file and the
|
|
953
|
+
``"<pid> <identity>"`` form (see :func:`read_pid_identity`)."""
|
|
954
|
+
return read_pid_identity(run_dir)[0]
|
|
955
|
+
|
|
956
|
+
|
|
957
|
+
def read_pid_identity(run_dir: Path) -> tuple[int | None, float | None]:
|
|
958
|
+
"""The recorded engine pid and its persisted identity, from ``<run_dir>/engine.pid``.
|
|
959
|
+
Thin wrapper over :func:`read_named_pid_identity` (which other pid files — the
|
|
960
|
+
Unity dialog probe's — reuse)."""
|
|
961
|
+
return read_named_pid_identity(run_dir / PID_FILE)
|
|
962
|
+
|
|
963
|
+
|
|
964
|
+
def read_named_pid_identity(pidfile: Path) -> tuple[int | None, float | None]:
|
|
965
|
+
"""The pid and its persisted identity recorded in ``pidfile``. ``(None, None)``
|
|
966
|
+
when the file is missing or the pid is unparseable; identity ``None`` for a legacy
|
|
967
|
+
pid-only file (callers then degrade to a bare existence check). A malformed
|
|
968
|
+
second token is not legacy: it returns an impossible identity so reuse guards
|
|
969
|
+
fail closed. First token is the pid, an optional second token the identity float."""
|
|
970
|
+
try:
|
|
971
|
+
tokens = pidfile.read_text(encoding="utf-8").split()
|
|
972
|
+
except OSError:
|
|
973
|
+
return None, None
|
|
974
|
+
if not tokens:
|
|
975
|
+
return None, None
|
|
976
|
+
try:
|
|
977
|
+
pid = int(tokens[0])
|
|
978
|
+
except ValueError:
|
|
979
|
+
return None, None
|
|
980
|
+
identity: float | None = None
|
|
981
|
+
if len(tokens) > 1:
|
|
982
|
+
try:
|
|
983
|
+
parsed = float(tokens[1])
|
|
984
|
+
except ValueError:
|
|
985
|
+
parsed = _INVALID_PID_IDENTITY
|
|
986
|
+
# Only a true one-token legacy file degrades to bare existence. If an
|
|
987
|
+
# identity token is present but corrupt/non-finite, fail closed as not-ours.
|
|
988
|
+
identity = parsed if math.isfinite(parsed) else _INVALID_PID_IDENTITY
|
|
989
|
+
return pid, identity
|
|
990
|
+
|
|
991
|
+
|
|
992
|
+
def engine_alive(run_dir: Path) -> bool:
|
|
993
|
+
"""True only when a local engine pid is provably alive **and still our engine**
|
|
994
|
+
(identity-checked, so a reused pid reads as dead). Mirrors :func:`liveness`
|
|
995
|
+
minus the tmux fallback — callers here want a definite 'is something running'
|
|
996
|
+
answer, and 'unknown' must not block stop/delete."""
|
|
997
|
+
pid, identity = read_pid_identity(run_dir)
|
|
998
|
+
if pid is None:
|
|
999
|
+
return False
|
|
1000
|
+
return get_process_host().alive_and_ours(pid, identity)
|
|
1001
|
+
|
|
1002
|
+
|
|
1003
|
+
def engine_liveness(run_dir: Path) -> str:
|
|
1004
|
+
"""Tri-state read of the local engine: ``'alive'`` | ``'dead'`` | ``'unknown'``.
|
|
1005
|
+
Wraps :meth:`ProcessHost.liveness_of` so a live-but-unreadable pid (win32
|
|
1006
|
+
``ERROR_ACCESS_DENIED``) reads ``'unknown'``, not a false ``'dead'``. No pid →
|
|
1007
|
+
``'dead'`` (the session fallback lives in the TUI layer)."""
|
|
1008
|
+
pid, identity = read_pid_identity(run_dir)
|
|
1009
|
+
if pid is None:
|
|
1010
|
+
return "dead"
|
|
1011
|
+
return probe_liveness(pid, identity)
|
|
1012
|
+
|
|
1013
|
+
|
|
1014
|
+
def probe_liveness(pid: int, identity: float | None) -> str:
|
|
1015
|
+
"""Tri-state probe of an already-read ``(pid, identity)`` — the shared body of
|
|
1016
|
+
:func:`engine_liveness` and :func:`liveness`, so both read the pid file once.
|
|
1017
|
+
A probe failure degrades to ``'unknown'``, never a false ``'dead'``."""
|
|
1018
|
+
host = get_process_host() # ProcessHostError (misconfig) propagates, not masked as unknown
|
|
1019
|
+
try:
|
|
1020
|
+
return host.liveness_of(pid, identity)
|
|
1021
|
+
except Exception:
|
|
1022
|
+
return "unknown"
|
|
1023
|
+
|
|
1024
|
+
|
|
1025
|
+
# ------------------------------------------------- run inventory / classification
|
|
1026
|
+
|
|
1027
|
+
|
|
1028
|
+
# Run statuses reported by `froid-loop list` and the dashboard.
|
|
1029
|
+
RUNNING = "running"
|
|
1030
|
+
PAUSED = "paused"
|
|
1031
|
+
FINISHED = "finished"
|
|
1032
|
+
STOPPED = "stopped"
|
|
1033
|
+
CRASHED = "crashed"
|
|
1034
|
+
INTERRUPTED = "interrupted"
|
|
1035
|
+
UNKNOWN = "unknown"
|
|
1036
|
+
|
|
1037
|
+
_StatSig = tuple[int, int, int]
|
|
1038
|
+
|
|
1039
|
+
|
|
1040
|
+
def _stat_sig(path: Path) -> _StatSig | None:
|
|
1041
|
+
try:
|
|
1042
|
+
st = path.stat()
|
|
1043
|
+
except OSError:
|
|
1044
|
+
return None
|
|
1045
|
+
# st_ino joins (mtime_ns, size): the engine rewrites state.json atomically
|
|
1046
|
+
# (temp + os.replace), so every write lands on a fresh inode. That catches a
|
|
1047
|
+
# same-size rewrite within one coarse mtime tick (e.g. WSL2 drvfs, or any fast
|
|
1048
|
+
# rewrite on a low-resolution mtime) that (mtime_ns, size) alone would miss and
|
|
1049
|
+
# serve stale from cache.
|
|
1050
|
+
return (st.st_mtime_ns, st.st_size, st.st_ino)
|
|
1051
|
+
|
|
1052
|
+
|
|
1053
|
+
def liveness(run_dir: Path) -> str:
|
|
1054
|
+
"""'alive' | 'dead' | 'unknown' for the engine that owns run_dir.
|
|
1055
|
+
|
|
1056
|
+
engine.pid is authoritative (written at run/sweep/resume start, never
|
|
1057
|
+
deleted). Legacy runs without one fall back to the per-run agent session —
|
|
1058
|
+
but that session only exists while an agent session runs, so its absence
|
|
1059
|
+
proves nothing: 'unknown', never falsely dead. Pid checks are local-only;
|
|
1060
|
+
runs on other hosts always come back 'unknown'.
|
|
1061
|
+
"""
|
|
1062
|
+
pid, identity = read_pid_identity(run_dir)
|
|
1063
|
+
if pid is None:
|
|
1064
|
+
return _session_liveness(run_dir.name)
|
|
1065
|
+
# Probe the pid we just read (shared body with engine_liveness) rather than
|
|
1066
|
+
# re-reading it, so a non-atomic pid rewrite can't split the two reads and flash
|
|
1067
|
+
# a false 'dead' between "pid present" here and a re-read seeing an empty file.
|
|
1068
|
+
try:
|
|
1069
|
+
return probe_liveness(pid, identity)
|
|
1070
|
+
except ProcessHostError:
|
|
1071
|
+
# A misconfigured host (bad FROID_LOOP_PROCESS_HOST) stays a hard error on
|
|
1072
|
+
# CLI decision paths, but the display layer must degrade, not crash: the
|
|
1073
|
+
# dashboard poll worker has no except and would take the whole app down.
|
|
1074
|
+
return "unknown"
|
|
1075
|
+
|
|
1076
|
+
|
|
1077
|
+
def _session_liveness(run_id: str) -> str:
|
|
1078
|
+
# An absent multiplexer / dead query proves nothing about a legacy run, so the
|
|
1079
|
+
# only positive signal is a live session; everything else is 'unknown'.
|
|
1080
|
+
mux = get_multiplexer()
|
|
1081
|
+
if not mux_usable(mux): # forced-aware, like every other observer gate
|
|
1082
|
+
return "unknown"
|
|
1083
|
+
try:
|
|
1084
|
+
return "alive" if mux.has_session(session_name(run_id)) else "unknown"
|
|
1085
|
+
except (OSError, MultiplexerError):
|
|
1086
|
+
# The seam raises MultiplexerError (not OSError) on a backend failure; a
|
|
1087
|
+
# dead query proves nothing about a legacy run, so degrade to 'unknown'
|
|
1088
|
+
# rather than crashing the TUI poll.
|
|
1089
|
+
return "unknown"
|
|
1090
|
+
|
|
1091
|
+
|
|
1092
|
+
def _classify(finished: bool, paused: bool, stopped: bool, crashed: bool, run_dir: Path) -> str:
|
|
1093
|
+
if finished:
|
|
1094
|
+
return FINISHED
|
|
1095
|
+
if paused:
|
|
1096
|
+
return PAUSED
|
|
1097
|
+
# a deliberate stop leaves a dead pid — check it before liveness so it does
|
|
1098
|
+
# not read as INTERRUPTED (a crash).
|
|
1099
|
+
if stopped:
|
|
1100
|
+
return STOPPED
|
|
1101
|
+
# a recorded crash leaves a dead pid too — surface it as a distinct CRASHED
|
|
1102
|
+
# before liveness, where it would otherwise read as a generic INTERRUPTED.
|
|
1103
|
+
if crashed:
|
|
1104
|
+
return CRASHED
|
|
1105
|
+
live = liveness(run_dir)
|
|
1106
|
+
if live == "alive":
|
|
1107
|
+
return RUNNING
|
|
1108
|
+
if live == "dead":
|
|
1109
|
+
return INTERRUPTED
|
|
1110
|
+
return UNKNOWN
|
|
1111
|
+
|
|
1112
|
+
|
|
1113
|
+
@dataclass(frozen=True)
|
|
1114
|
+
class RunInfo:
|
|
1115
|
+
run_id: str
|
|
1116
|
+
run_dir: Path
|
|
1117
|
+
run_type: str
|
|
1118
|
+
started_at: str
|
|
1119
|
+
status: str
|
|
1120
|
+
paused_stage: str = "" # RunState.paused_stage when PAUSED, else ""; drives the badge
|
|
1121
|
+
stopping: bool = False # a graceful stop is pending (control file present) while RUNNING
|
|
1122
|
+
|
|
1123
|
+
|
|
1124
|
+
# state.json path -> (stat sig, header fields tuple)
|
|
1125
|
+
_HeaderFields = tuple[str, str, bool, bool, bool, bool, str]
|
|
1126
|
+
_header_cache: dict[Path, tuple[_StatSig, _HeaderFields]] = {}
|
|
1127
|
+
|
|
1128
|
+
|
|
1129
|
+
def discover_runs(project: Path) -> list[RunInfo]:
|
|
1130
|
+
"""One RunInfo per run dir, oldest first; [] when the runs dir is missing.
|
|
1131
|
+
|
|
1132
|
+
Parses only the state.json header fields (cached on stat); a state file
|
|
1133
|
+
that fails to parse yields status 'unknown' rather than crashing — it is
|
|
1134
|
+
transient, the engine writes atomically.
|
|
1135
|
+
"""
|
|
1136
|
+
out: list[RunInfo] = []
|
|
1137
|
+
for run_dir in list_run_dirs(project):
|
|
1138
|
+
state_path = run_dir / STATE_FILE
|
|
1139
|
+
sig = _stat_sig(state_path)
|
|
1140
|
+
cached = _header_cache.get(state_path)
|
|
1141
|
+
if sig is not None and cached is not None and cached[0] == sig:
|
|
1142
|
+
run_type, started_at, finished, paused, stopped, crashed, paused_stage = cached[1]
|
|
1143
|
+
else:
|
|
1144
|
+
try:
|
|
1145
|
+
doc = json.loads(state_path.read_text(encoding="utf-8"))
|
|
1146
|
+
run_type = str(doc.get("run_type", "story"))
|
|
1147
|
+
started_at = str(doc.get("started_at", ""))
|
|
1148
|
+
finished = bool(doc.get("finished", False))
|
|
1149
|
+
paused = doc.get("paused_reason") is not None
|
|
1150
|
+
stopped = bool(doc.get("stopped", False))
|
|
1151
|
+
crashed = bool(doc.get("crashed", False))
|
|
1152
|
+
paused_stage = str(doc.get("paused_stage") or "")
|
|
1153
|
+
except (OSError, json.JSONDecodeError):
|
|
1154
|
+
out.append(RunInfo(run_dir.name, run_dir, "?", "", UNKNOWN))
|
|
1155
|
+
continue
|
|
1156
|
+
if sig is not None:
|
|
1157
|
+
_header_cache[state_path] = (
|
|
1158
|
+
sig,
|
|
1159
|
+
(run_type, started_at, finished, paused, stopped, crashed, paused_stage),
|
|
1160
|
+
)
|
|
1161
|
+
status = _classify(finished, paused, stopped, crashed, run_dir)
|
|
1162
|
+
# paused_stage is advisory: only meaningful while the run is actually PAUSED
|
|
1163
|
+
# (a resumed run keeps the last stage in state until it re-pauses/finishes).
|
|
1164
|
+
stage = paused_stage if status == PAUSED else ""
|
|
1165
|
+
# A pending graceful stop is the control file's presence, but only while an
|
|
1166
|
+
# engine is still around to honor it — RUNNING or UNKNOWN (an unverifiable
|
|
1167
|
+
# pid still consumes the file). The engine discards the file at the stop
|
|
1168
|
+
# boundary, so a lingering file on an already-concluded run is not "stopping":
|
|
1169
|
+
# STOPPED/FINISHED/CRASHED classify before liveness, so they never read UNKNOWN.
|
|
1170
|
+
stopping = status in (RUNNING, UNKNOWN) and (run_dir / STOP_REQUEST_FILE).is_file()
|
|
1171
|
+
out.append(RunInfo(run_dir.name, run_dir, run_type, started_at, status, stage, stopping))
|
|
1172
|
+
return out
|
|
1173
|
+
|
|
1174
|
+
|
|
1175
|
+
# ----------------------------------------------------------- stop / delete / archive
|
|
1176
|
+
|
|
1177
|
+
|
|
1178
|
+
def kill_session(run_id: str, mux: TerminalMultiplexer | None = None) -> None:
|
|
1179
|
+
"""Kill a run's agent session (froid-loop-<id>); a no-op when it is already
|
|
1180
|
+
gone or the multiplexer is unavailable.
|
|
1181
|
+
|
|
1182
|
+
Also a no-op for an id that **aliases a control session**
|
|
1183
|
+
(:func:`run_id_aliases_control_session` — ``ctl`` or ``ctl-<16 hex>``,
|
|
1184
|
+
case-folded): the only session such a name can address is the control
|
|
1185
|
+
plane, every parked window of every run in it. Unreachable through
|
|
1186
|
+
minting (validation refuses the shape) but reachable through what an
|
|
1187
|
+
**older release persisted**: a run dir named ``ctl`` that `stop`,
|
|
1188
|
+
`delete` or a resume's stale-session sweep replays as a kill target.
|
|
1189
|
+
This chokepoint keeps those read paths safe — and usable as the
|
|
1190
|
+
operator's way out of such a run — without each caller re-deriving the
|
|
1191
|
+
rule.
|
|
1192
|
+
|
|
1193
|
+
The narrow test, not the mint's broad reservation, deliberately: a
|
|
1194
|
+
historical ``ctl-foo`` run DOES own an agent session of its own
|
|
1195
|
+
(``froid-loop-ctl-foo``, distinct from every control session and killed
|
|
1196
|
+
exactly — the seam sends ``=``-exact tmux targets, and psmux resolves
|
|
1197
|
+
exact port files), and skipping its kill stranded it: the prune already
|
|
1198
|
+
could not reach it, so nothing could. Scope, stated: the kill addresses
|
|
1199
|
+
the registry THIS process addresses — a pre-upgrade session left in
|
|
1200
|
+
psmux's old default registry is not reachable from here (measured), and
|
|
1201
|
+
deliberately so: a by-name kill in a shared registry without tag proof
|
|
1202
|
+
could take another project's same-named session (run ids are unique per
|
|
1203
|
+
project only). The legacy sweep in :func:`prune_sessions`, which does
|
|
1204
|
+
demand the tag, is the path that reaches it."""
|
|
1205
|
+
if run_id_aliases_control_session(run_id):
|
|
1206
|
+
return
|
|
1207
|
+
(mux or get_multiplexer()).kill_session(session_name(run_id))
|
|
1208
|
+
|
|
1209
|
+
|
|
1210
|
+
CTL_SESSION = "froid-loop-ctl"
|
|
1211
|
+
_SESSION_PREFIX = "froid-loop-"
|
|
1212
|
+
|
|
1213
|
+
|
|
1214
|
+
def ctl_session_for(project: Path, mux: TerminalMultiplexer | None = None) -> str:
|
|
1215
|
+
"""The control-session name this project's launches and lookups share.
|
|
1216
|
+
|
|
1217
|
+
On a transport with no registry namespace (tmux) it is the fixed
|
|
1218
|
+
:data:`CTL_SESSION`, machine-shared as it has always been. On a namespacing
|
|
1219
|
+
transport (psmux) the name carries the registry's identity — a 16-hex
|
|
1220
|
+
digest of the derived registry root — because the two scopes genuinely
|
|
1221
|
+
differ: the session lives *per registry*, but psmux's duplicate-server
|
|
1222
|
+
guard is a mutex keyed on the session name alone, across every registry
|
|
1223
|
+
in the **login session** (``Local\\psmux-session-{name}`` over
|
|
1224
|
+
``port_file_base()`` — the ``Local\\`` kernel-object namespace is
|
|
1225
|
+
per-login-session, not machine-global; ``server/mod.rs:853`` /
|
|
1226
|
+
``platform.rs:346`` / ``types.rs:1345``, source-read at v3.3.8 —
|
|
1227
|
+
``PSMUX_DATA_DIR`` never enters it). A fixed name therefore admits ONE
|
|
1228
|
+
control session across every registry a desktop session can reach,
|
|
1229
|
+
and the second project's create is rejected as a duplicate server — its
|
|
1230
|
+
TUI launch fails instead of minting its own session (measured: a second
|
|
1231
|
+
registry answers ``new-session`` rc 1 for the fixed name while the first
|
|
1232
|
+
registry's server lives, and rc 0 for a per-registry name).
|
|
1233
|
+
|
|
1234
|
+
The digest is over ``mux_registry_root(project)`` **resolved**: the name
|
|
1235
|
+
must be unique per *physical* registry, and the resolved path is that
|
|
1236
|
+
registry's identity — (project, state root), both axes; ``project_tag``
|
|
1237
|
+
alone would recreate the collision for one project under two state roots.
|
|
1238
|
+
Resolved rather than as spelled because two spellings of one state root
|
|
1239
|
+
(``C:\\work\\state`` vs ``C:\\work\\alias\\..\\state``) reach **one**
|
|
1240
|
+
registry — Windows resolves both to the same files, and psmux keeps the
|
|
1241
|
+
spelling only while constructing those paths (``src/paths.rs:79``,
|
|
1242
|
+
source-read at v3.3.8; convergence measured) — so an as-spelled digest
|
|
1243
|
+
minted two control sessions inside one registry, each blind to the other's
|
|
1244
|
+
parked windows: the split-control-plane failure again, one level up. Same
|
|
1245
|
+
rule ``project_tag`` already states: resolve *before* digesting.
|
|
1246
|
+
|
|
1247
|
+
…and then ``os.path.normcase``, because ``resolve()`` can only return the
|
|
1248
|
+
filesystem's stored case for a path that **exists**, and the registry
|
|
1249
|
+
root usually does not yet exist at the moment the name is needed (psmux
|
|
1250
|
+
``create_dir_all``\\s it at first spawn). Two case spellings of a
|
|
1251
|
+
not-yet-created state root resolve to two strings, digest to two names —
|
|
1252
|
+
and then land in ONE physical registry, because NTFS folds case when
|
|
1253
|
+
psmux opens the ``.port`` files (measured). ``normcase`` folds exactly
|
|
1254
|
+
where the filesystem does: it lowercases on Windows and is the identity
|
|
1255
|
+
on POSIX, where case is significant and two case spellings ARE two
|
|
1256
|
+
registries — folding there would merge genuinely distinct roots.
|
|
1257
|
+
Ceiling, named: ``normcase`` lowercases with ``str.lower``, which can
|
|
1258
|
+
disagree with NTFS's own fold table for a few non-ASCII case pairs; a
|
|
1259
|
+
state root spelled in two such casings of the same non-ASCII name stays
|
|
1260
|
+
split, as it is for every other digest of an operator-supplied path.
|
|
1261
|
+
|
|
1262
|
+
The degrade arm (namespaced transport, underivable state root) answers
|
|
1263
|
+
the fixed name: that arm runs on the transport's shared default registry,
|
|
1264
|
+
where a shared session scoped by per-window project tags is the correct,
|
|
1265
|
+
tmux-shaped semantic — and where a pre-#537 legacy ctl session under the
|
|
1266
|
+
fixed name may exist to be reused rather than collided with.
|
|
1267
|
+
"""
|
|
1268
|
+
mux = mux or get_multiplexer()
|
|
1269
|
+
if not mux.has_registry_namespace():
|
|
1270
|
+
return CTL_SESSION
|
|
1271
|
+
try:
|
|
1272
|
+
scope = os.path.normcase(str(mux_registry_root(project).resolve()))
|
|
1273
|
+
except (StateRootError, OSError, RuntimeError):
|
|
1274
|
+
return CTL_SESSION
|
|
1275
|
+
return f"{CTL_SESSION}-{hashlib.sha256(os.fsencode(scope)).hexdigest()[:16]}"
|
|
1276
|
+
|
|
1277
|
+
|
|
1278
|
+
def is_ctl_session_name(name: str) -> bool:
|
|
1279
|
+
"""Whether ``name`` is a control session's name — the fixed
|
|
1280
|
+
:data:`CTL_SESSION`, or ``froid-loop-ctl-<16 hex>``, the ONE suffix shape
|
|
1281
|
+
:func:`ctl_session_for` can mint.
|
|
1282
|
+
|
|
1283
|
+
The shape predicate exists because several readers ask "is this A control
|
|
1284
|
+
session" without a project in hand: the agent-session parser must exclude
|
|
1285
|
+
ctl sessions (``froid-loop-ctl-<16hex>`` would otherwise parse as run id
|
|
1286
|
+
``ctl-<16hex>``, which ``RUN_ID_RE`` admits), the legacy-leftovers reader
|
|
1287
|
+
names a surviving ctl session in a registry this process did not derive,
|
|
1288
|
+
and ``in_ctl_session`` classifies whatever session this process woke up
|
|
1289
|
+
inside.
|
|
1290
|
+
|
|
1291
|
+
Exactly the mintable shapes, no wider: an arbitrary suffix
|
|
1292
|
+
(``froid-loop-ctl-foo``) is NOT a control session — it is the agent
|
|
1293
|
+
session of a run an older release accepted as ``--run-id ctl-foo``, and
|
|
1294
|
+
reading it as a control session made it unreachable by ``stop`` and the
|
|
1295
|
+
prune both. No agent session of OURS can match this predicate:
|
|
1296
|
+
:func:`is_valid_run_id` refuses every ctl-shaped id at the mint (broad —
|
|
1297
|
+
:func:`is_reserved_run_id`), so a matching name is either genuinely a
|
|
1298
|
+
control session or hand-made to look like one — and the hand-made
|
|
1299
|
+
16-hex-suffixed case stays unprunable, the leak direction."""
|
|
1300
|
+
if name == CTL_SESSION:
|
|
1301
|
+
return True
|
|
1302
|
+
suffix = name.removeprefix(CTL_SESSION + "-")
|
|
1303
|
+
return suffix != name and len(suffix) == 16 and all(c in "0123456789abcdef" for c in suffix)
|
|
1304
|
+
|
|
1305
|
+
|
|
1306
|
+
# tmux user option stamping a session/window with the project it belongs to, so
|
|
1307
|
+
# a prune in one project never touches another project's live runs. See
|
|
1308
|
+
# prunable_sessions and tui.launch.
|
|
1309
|
+
PROJECT_OPTION = "@froid_project"
|
|
1310
|
+
|
|
1311
|
+
|
|
1312
|
+
def project_tag(project: Path) -> str:
|
|
1313
|
+
"""Canonical project identity used by both tag writers and prune readers. The
|
|
1314
|
+
single source of normalization: both sides must route through this so symlinks
|
|
1315
|
+
and relative paths can't make a project look foreign to its own sessions.
|
|
1316
|
+
|
|
1317
|
+
Hashing the resolved path makes every value safe by construction, on both
|
|
1318
|
+
transports a tag has to cross. It clears psmux's control line (#419), whose
|
|
1319
|
+
gate refuses any value the CLI->server hop would mangle — a UNC share whose
|
|
1320
|
+
name holds a space is refused verbatim, and that refusal left the session
|
|
1321
|
+
untagged, which is weak ownership twice over. It equally clears the listing
|
|
1322
|
+
round trip (#518): a hex digest holds nothing `str.splitlines()` breaks on,
|
|
1323
|
+
no tab, and no byte outside ASCII, so it can neither split a row nor fail the
|
|
1324
|
+
backends' strict decode.
|
|
1325
|
+
|
|
1326
|
+
That subsumes the conditional percent-encoding this function briefly applied.
|
|
1327
|
+
Encoding answered only the listing half, so a path the listing could carry but
|
|
1328
|
+
the control line could not — the spaced UNC above — still went untagged. The
|
|
1329
|
+
compatibility objection encoding was shaped around, that rewriting every tag
|
|
1330
|
+
strands the ones already stored on live sessions and windows, is answered on
|
|
1331
|
+
the read side instead, by `accepted_tags`.
|
|
1332
|
+
|
|
1333
|
+
16 hex characters are ample for one machine's project population.
|
|
1334
|
+
"""
|
|
1335
|
+
return hashlib.sha256(os.fsencode(str(project.resolve()))).hexdigest()[:16]
|
|
1336
|
+
|
|
1337
|
+
|
|
1338
|
+
def accepted_tags(project: Path) -> frozenset[str]:
|
|
1339
|
+
"""Current digest plus the legacy resolved-path tag accepted during pruning.
|
|
1340
|
+
|
|
1341
|
+
The legacy member is read-only compatibility for sessions and ctl windows that
|
|
1342
|
+
survive an upgrade; remove it once no path-tagged multiplexer state can remain.
|
|
1343
|
+
Returns the whole set rather than answering per tag so a read site resolves the
|
|
1344
|
+
project once per prune instead of once per session.
|
|
1345
|
+
|
|
1346
|
+
The two shapes cannot collide into false ownership: a legacy tag is an absolute
|
|
1347
|
+
path, so it always holds a separator, while a digest is bare 16-hex.
|
|
1348
|
+
|
|
1349
|
+
Deliberately two members and not three — a tag spelled with the `%enc%` prefix,
|
|
1350
|
+
from the window when this module encoded rather than hashed, is not accepted.
|
|
1351
|
+
Only a path the listing could not carry was ever spelled that way (one holding
|
|
1352
|
+
a line separator, or a byte invalid in the filesystem encoding), and that
|
|
1353
|
+
spelling never reached a release. An unaccepted tag reads as foreign, which
|
|
1354
|
+
skips the session rather than pruning it, so the edge is fail-safe and clears
|
|
1355
|
+
itself on the next tag write.
|
|
1356
|
+
"""
|
|
1357
|
+
return frozenset({project_tag(project), str(project.resolve())})
|
|
1358
|
+
|
|
1359
|
+
|
|
1360
|
+
def lock_path_for(data_path: Path) -> Path:
|
|
1361
|
+
"""The advisory-lock sidecar for a mutable data file:
|
|
1362
|
+
``<state root>/locks/<sha256(resolved path)[:16]>-<basename>.lock``.
|
|
1363
|
+
|
|
1364
|
+
Out of the repository, deliberately, and NOT the ``<file>.lock`` sibling the
|
|
1365
|
+
obvious reading of #286 asks for. The deferred-work ledger is a *tracked*
|
|
1366
|
+
file by design, and both :func:`verify.commit_story` and
|
|
1367
|
+
:func:`verify.finalize_commit` stage with ``git add -A``: a lock beside it
|
|
1368
|
+
would be swept into the engine's own commits, and the git-add shield that
|
|
1369
|
+
would otherwise hide it covers linked worktrees only. Under the state root
|
|
1370
|
+
the sidecar is never git-visible at all, so no exclusion machinery has to be
|
|
1371
|
+
kept correct for it.
|
|
1372
|
+
|
|
1373
|
+
Keyed on the **resolved** path so the identity of the lock is the identity of
|
|
1374
|
+
the file rather than of the spelling used to reach it: a symlinked and a
|
|
1375
|
+
direct path to one ledger rendezvous on one lock (without which the two
|
|
1376
|
+
spellings would exclude nobody), two worktrees' in-tree ledgers are different
|
|
1377
|
+
files and correctly get independent locks, and several projects pointed at a
|
|
1378
|
+
shared external artifact dir land on one lock, which is where the real
|
|
1379
|
+
contention is. The basename is appended for debuggability only — a human
|
|
1380
|
+
reading ``ls`` of the locks dir should see which file a sidecar guards — and
|
|
1381
|
+
carries no meaning for exclusion, which rides the digest.
|
|
1382
|
+
|
|
1383
|
+
Pure: no directory is created here, because
|
|
1384
|
+
:func:`~froid_loop.platform_util.file_lock` mkdirs the parent when it opens
|
|
1385
|
+
the lock. May raise :class:`StateRootError` when the environment names no
|
|
1386
|
+
usable state root (see :func:`state_root`); the caller fails rather than
|
|
1387
|
+
silently locking somewhere else.
|
|
1388
|
+
"""
|
|
1389
|
+
resolved = data_path.resolve()
|
|
1390
|
+
digest = hashlib.sha256(os.fsencode(str(resolved))).hexdigest()[:16]
|
|
1391
|
+
return state_root() / "locks" / f"{digest}-{resolved.name}.lock"
|
|
1392
|
+
|
|
1393
|
+
|
|
1394
|
+
def mux_sessions() -> list[str]:
|
|
1395
|
+
"""All live session names, or [] when the multiplexer is missing, no server
|
|
1396
|
+
is running, or the query fails."""
|
|
1397
|
+
return get_multiplexer().list_sessions()
|
|
1398
|
+
|
|
1399
|
+
|
|
1400
|
+
def session_project_tags() -> dict[str, str]:
|
|
1401
|
+
"""Map each live session name to its PROJECT_OPTION value ("" when unset).
|
|
1402
|
+
Same missing-multiplexer/no-server guards as mux_sessions()."""
|
|
1403
|
+
return get_multiplexer().session_options(PROJECT_OPTION)
|
|
1404
|
+
|
|
1405
|
+
|
|
1406
|
+
def _agent_run_id(session: str) -> str | None:
|
|
1407
|
+
"""The run id behind a ``froid-loop-<id>`` agent session name, or ``None`` when
|
|
1408
|
+
the name is not one: the control session, a foreign session, or a mangled name
|
|
1409
|
+
whose id could not be replayed as a path segment — never let one steer a
|
|
1410
|
+
run-dir path. Shared so the prune partition and
|
|
1411
|
+
:func:`legacy_registry_leftovers` cannot drift on what counts as ours.
|
|
1412
|
+
|
|
1413
|
+
The id question is :func:`is_parsable_run_id`, deliberately NOT
|
|
1414
|
+
:func:`is_valid_run_id` — the parse side must accept ids the mint refuses.
|
|
1415
|
+
That predicate owns the reasoning, and the ctl-window sweep in
|
|
1416
|
+
``tui.launch`` asks the same one."""
|
|
1417
|
+
if not session.startswith(_SESSION_PREFIX):
|
|
1418
|
+
return None
|
|
1419
|
+
run_id = session[len(_SESSION_PREFIX) :]
|
|
1420
|
+
return run_id if is_parsable_run_id(run_id) else None
|
|
1421
|
+
|
|
1422
|
+
|
|
1423
|
+
def prunable_sessions(
|
|
1424
|
+
project: Path, mux: TerminalMultiplexer | None = None, *, require_tag: bool = False
|
|
1425
|
+
) -> tuple[list[str], list[str], set[str]]:
|
|
1426
|
+
"""Partition the froid-loop-<id> agent sessions into (prunable, live) run ids,
|
|
1427
|
+
plus the subset of prunable ids whose engine liveness read 'unknown'
|
|
1428
|
+
(unverifiable pid). Unknown never blocks cleanup — those sessions stay
|
|
1429
|
+
prunable — but frontends surface a warning for them.
|
|
1430
|
+
|
|
1431
|
+
A control session (:func:`is_ctl_session_name` — the fixed name or a
|
|
1432
|
+
per-registry one) is never a candidate. Pruning is scoped
|
|
1433
|
+
to `project` via the PROJECT_OPTION tag set at session creation:
|
|
1434
|
+
|
|
1435
|
+
- tag proves this project (see accepted_tags): ours — prunable unless a
|
|
1436
|
+
provably-alive engine pid is running (covers finished/stopped/crashed *and*
|
|
1437
|
+
orphans whose run dir was deleted, since engine_liveness reads 'dead' with
|
|
1438
|
+
no pid).
|
|
1439
|
+
- tag is another project: skipped — never touched.
|
|
1440
|
+
- tag empty (untagged session): can't prove ownership, so fall back to the run
|
|
1441
|
+
dir — prunable only when the dir exists under this project and is dead;
|
|
1442
|
+
skipped when the dir is absent. Reachable when the tag write failed, when
|
|
1443
|
+
the option read degrades (session_options reads unset as "no answer", never
|
|
1444
|
+
as proof nothing was written), or on a session predating a working tag
|
|
1445
|
+
write — e.g. psmux path tags refused before the digest.
|
|
1446
|
+
|
|
1447
|
+
``require_tag`` drops that last arm: an untagged session is skipped outright
|
|
1448
|
+
rather than falling back to the run dir. Set for a **shared** registry — the
|
|
1449
|
+
legacy psmux root every project's pre-upgrade sessions sit in together (see
|
|
1450
|
+
:func:`prune_sessions`). The fallback proves ownership from
|
|
1451
|
+
``run_dir_for(project, run_id)``, and a run id is only unique *within* one
|
|
1452
|
+
project (``--run-id`` is caller-supplied), so in a shared registry a dead run
|
|
1453
|
+
dir here is not evidence about a session over there: project A holding a dead
|
|
1454
|
+
`shared-id` would claim project B's live, untagged `froid-loop-shared-id` and
|
|
1455
|
+
kill it. In a per-project registry the same fallback is sound because the
|
|
1456
|
+
registry itself proves ownership, which is why the flag is off by default and
|
|
1457
|
+
the primary pass keeps the reach it always had. What the flag leaves behind is
|
|
1458
|
+
reported by :func:`legacy_registry_leftovers`.
|
|
1459
|
+
"""
|
|
1460
|
+
# `mux` bypasses the module-level readers rather than widening them: those
|
|
1461
|
+
# two are the seam every other caller (and every test) reaches the process-wide
|
|
1462
|
+
# backend through, and a bound instance is this function's business alone.
|
|
1463
|
+
tags = mux.session_options(PROJECT_OPTION) if mux is not None else session_project_tags()
|
|
1464
|
+
mine = accepted_tags(project)
|
|
1465
|
+
prunable: list[str] = []
|
|
1466
|
+
live: list[str] = []
|
|
1467
|
+
unknown: set[str] = set()
|
|
1468
|
+
names = mux.list_sessions() if mux is not None else mux_sessions()
|
|
1469
|
+
for name in names:
|
|
1470
|
+
run_id = _agent_run_id(name)
|
|
1471
|
+
if run_id is None:
|
|
1472
|
+
continue
|
|
1473
|
+
run_dir = run_dir_for(project, run_id)
|
|
1474
|
+
tag = tags.get(name, "")
|
|
1475
|
+
if tag:
|
|
1476
|
+
if tag not in mine:
|
|
1477
|
+
continue # another project's session
|
|
1478
|
+
elif require_tag or not is_run(run_dir):
|
|
1479
|
+
continue # ownership unprovable: no tag, and no run dir here to stand in
|
|
1480
|
+
liveness = engine_liveness(run_dir)
|
|
1481
|
+
if liveness == "alive":
|
|
1482
|
+
live.append(run_id)
|
|
1483
|
+
continue
|
|
1484
|
+
prunable.append(run_id)
|
|
1485
|
+
if liveness == "unknown":
|
|
1486
|
+
unknown.add(run_id)
|
|
1487
|
+
return prunable, live, unknown
|
|
1488
|
+
|
|
1489
|
+
|
|
1490
|
+
def _registry_proves_ownership(project: Path) -> bool:
|
|
1491
|
+
"""True when the registry the *primary* prune pass addresses is one froid-loop
|
|
1492
|
+
derived for this project — which is what makes
|
|
1493
|
+
:func:`prunable_sessions`' untagged run-dir fallback evidence rather than a
|
|
1494
|
+
guess.
|
|
1495
|
+
|
|
1496
|
+
That fallback claims an untagged ``froid-loop-<id>`` session when this project
|
|
1497
|
+
holds a dead run dir of the same id. Run ids are only unique *within* a
|
|
1498
|
+
project (``--run-id`` is caller-supplied), so the claim is sound exactly when
|
|
1499
|
+
the registry itself already restricts what can be listed to this project's
|
|
1500
|
+
sessions. In a registry shared with other projects — or with the operator —
|
|
1501
|
+
it is not, and the same reasoning that put ``require_tag=True`` on the legacy
|
|
1502
|
+
pass applies here.
|
|
1503
|
+
|
|
1504
|
+
The primary registry is not always ours. :func:`export_psmux_registry_root`
|
|
1505
|
+
degrades to ``None`` on an underivable state root and leaves whatever ambient
|
|
1506
|
+
``PSMUX_DATA_DIR`` it found in force, and psmux honours any absolute value
|
|
1507
|
+
(``src/paths.rs``, source-read at v3.3.8) — so on that arm every verb,
|
|
1508
|
+
including the kill, addresses the operator's own registry while this project's
|
|
1509
|
+
run dirs go on looking like ownership.
|
|
1510
|
+
|
|
1511
|
+
``registry_root()`` answering ``None`` covers two cases, and they get
|
|
1512
|
+
**opposite** answers — conflating them was a defect, not caution. A backend
|
|
1513
|
+
with no registry namespace at all (tmux: one server for the machine,
|
|
1514
|
+
``has_registry_namespace()`` False) keeps the reach it had before
|
|
1515
|
+
per-project registries existed: the listing there is exactly what it always
|
|
1516
|
+
was, and narrowing it would be a regression dressed as caution. A backend
|
|
1517
|
+
that DOES namespace and has no root in force (psmux with ``PSMUX_DATA_DIR``
|
|
1518
|
+
unset — the export degraded on an underivable state root and there was no
|
|
1519
|
+
ambient value either) is running on its transport's own **default**
|
|
1520
|
+
registry — shared with every other project and with the operator
|
|
1521
|
+
(``<home>\\.psmux``, the home being ``USERPROFILE`` when set, else the
|
|
1522
|
+
profile API, else ``HOMEDRIVE``+``HOMEPATH``, else ``HOME`` —
|
|
1523
|
+
``src/paths.rs`` ``home_dir``, source-read at v3.3.8) — which proves nothing about
|
|
1524
|
+
ownership, exactly as an absolute ambient value naming a foreign registry
|
|
1525
|
+
proves nothing. Both shared cases make the tag mandatory.
|
|
1526
|
+
|
|
1527
|
+
A backend that cannot be asked answers ``False``: the safe direction is to
|
|
1528
|
+
demand the tag, which leaves a session standing rather than killing one on
|
|
1529
|
+
evidence that may not hold.
|
|
1530
|
+
"""
|
|
1531
|
+
try:
|
|
1532
|
+
mux = get_multiplexer()
|
|
1533
|
+
root = mux.registry_root()
|
|
1534
|
+
if root is None:
|
|
1535
|
+
# No namespace (tmux): historical reach. A namespace with no root
|
|
1536
|
+
# in force is the transport's shared default registry: demand the tag.
|
|
1537
|
+
return not mux.has_registry_namespace()
|
|
1538
|
+
except MultiplexerError:
|
|
1539
|
+
return False
|
|
1540
|
+
try:
|
|
1541
|
+
return root == str(mux_registry_root(project))
|
|
1542
|
+
except (StateRootError, OSError, RuntimeError):
|
|
1543
|
+
return False
|
|
1544
|
+
|
|
1545
|
+
|
|
1546
|
+
def prune_sessions(
|
|
1547
|
+
project: Path, *, dry_run: bool = False
|
|
1548
|
+
) -> tuple[list[str], list[str], set[str]]:
|
|
1549
|
+
"""Kill every prunable froid-loop-<id> session (see prunable_sessions);
|
|
1550
|
+
returns (killed, live, unknown): the run ids that were (or, with dry_run,
|
|
1551
|
+
would be) killed, the live ids skipped, and the killed subset whose engine
|
|
1552
|
+
liveness read 'unknown'. All three come from the same partition sample, so
|
|
1553
|
+
frontend messaging built from them always describes the performed actions.
|
|
1554
|
+
|
|
1555
|
+
Runs once per registry: the one this process is pointed at, then each legacy
|
|
1556
|
+
registry the backend still admits (:func:`_legacy_registries`). Sessions
|
|
1557
|
+
froid-loop created before it took a per-project psmux root are addressable
|
|
1558
|
+
only from the second pass, and without it cleanup would report a clean sweep
|
|
1559
|
+
while their servers ran on. The passes are unioned rather than concatenated —
|
|
1560
|
+
a run id can only be in one registry, but a backend answering the same
|
|
1561
|
+
registry twice must not make one kill look like two.
|
|
1562
|
+
|
|
1563
|
+
Ownership is judged per pass by the same :func:`prunable_sessions` partition,
|
|
1564
|
+
so a legacy registry buys no extra reach: another project's sessions and the
|
|
1565
|
+
operator's own psmux sessions are skipped there exactly as they are here.
|
|
1566
|
+
|
|
1567
|
+
The legacy pass always runs with ``require_tag=True``, and the primary pass
|
|
1568
|
+
runs with it whenever the registry it addresses is not one froid-loop derived
|
|
1569
|
+
for this project (:func:`_registry_proves_ownership`). Both are the same rule:
|
|
1570
|
+
:func:`prunable_sessions`' untagged run-dir fallback is evidence only where
|
|
1571
|
+
the registry has already restricted the listing to this project. A legacy
|
|
1572
|
+
registry is shared by every project by definition; the primary one is shared
|
|
1573
|
+
whenever the derivation failed — an ambient ``PSMUX_DATA_DIR`` left in
|
|
1574
|
+
force, or nothing in force at all, where a namespacing backend runs on its
|
|
1575
|
+
own shared default registry. What that strictness leaves standing in a
|
|
1576
|
+
legacy registry is reported
|
|
1577
|
+
by :func:`legacy_registry_leftovers`, which the cleanup frontends print: a
|
|
1578
|
+
sweep that silently declines to migrate something is the same silence this
|
|
1579
|
+
whole change exists to remove."""
|
|
1580
|
+
prunable, live, unknown = prunable_sessions(
|
|
1581
|
+
project, require_tag=not _registry_proves_ownership(project)
|
|
1582
|
+
)
|
|
1583
|
+
if not dry_run:
|
|
1584
|
+
for run_id in prunable:
|
|
1585
|
+
kill_session(run_id)
|
|
1586
|
+
for legacy in _legacy_registries():
|
|
1587
|
+
extra, extra_live, extra_unknown = prunable_sessions(project, legacy, require_tag=True)
|
|
1588
|
+
if not dry_run:
|
|
1589
|
+
for run_id in extra:
|
|
1590
|
+
kill_session(run_id, legacy)
|
|
1591
|
+
prunable += [i for i in extra if i not in prunable]
|
|
1592
|
+
live += [i for i in extra_live if i not in live]
|
|
1593
|
+
unknown |= extra_unknown
|
|
1594
|
+
return prunable, live, unknown
|
|
1595
|
+
|
|
1596
|
+
|
|
1597
|
+
#: How a frontend names psmux's OWN default registry, the one root
|
|
1598
|
+
#: :meth:`~.adapters.multiplexer.TerminalMultiplexer.registry_root` deliberately
|
|
1599
|
+
#: answers ``None`` for (respelling its home cascade in Python is a second thing
|
|
1600
|
+
#: to keep in sync). Lives here so both frontends say it the same way.
|
|
1601
|
+
DEFAULT_REGISTRY_LABEL = "the multiplexer's own default registry"
|
|
1602
|
+
|
|
1603
|
+
|
|
1604
|
+
def legacy_registry_leftovers(
|
|
1605
|
+
project: Path, *, announced: Iterable[str] = ()
|
|
1606
|
+
) -> dict[str, list[str]]:
|
|
1607
|
+
"""Session names a legacy registry **still holds** after :func:`prune_sessions`
|
|
1608
|
+
ran — the migration's honest remainder, for the cleanup frontends to print.
|
|
1609
|
+
``{}`` when there is no legacy registry, when they hold nothing, or when
|
|
1610
|
+
every listing fails.
|
|
1611
|
+
|
|
1612
|
+
**Grouped by registry, and that is load-bearing.** There is more than one
|
|
1613
|
+
legacy registry now (:meth:`~.adapters.psmux_backend.PsmuxMultiplexer.legacy_registries`
|
|
1614
|
+
— psmux's default, and the root this process displaced), so a flat list
|
|
1615
|
+
cannot say where to go look: a message built from one would either name a
|
|
1616
|
+
registry the leftovers are not in, or name every registry the sweep
|
|
1617
|
+
addressed including the ones that contributed nothing. The operator's next
|
|
1618
|
+
action is to open that registry, so the answer has to be per registry. Keys
|
|
1619
|
+
are :meth:`registry_root`'s answer, or :data:`DEFAULT_REGISTRY_LABEL` where
|
|
1620
|
+
that is ``None``; a registry holding nothing is absent rather than empty, so
|
|
1621
|
+
a caller can print the keys without checking.
|
|
1622
|
+
|
|
1623
|
+
**Presence, not a second opinion.** Called after the sweep, this lists what is
|
|
1624
|
+
actually there; a session the sweep killed is simply gone from the listing.
|
|
1625
|
+
That is the whole judgement for anything tagged as ours, and it is deliberately
|
|
1626
|
+
*not* a re-run of the partition: re-judging liveness would open a race the
|
|
1627
|
+
reader cannot see the far side of. A run alive during the prune (correctly
|
|
1628
|
+
left, and reported ``live``) can exit before the reader looks; a resampled
|
|
1629
|
+
partition would then call it ``prunable``, and it would fall out of both the
|
|
1630
|
+
live arm and the untagged fallback — stranded and unreported, with no kill ever
|
|
1631
|
+
attempted. Presence has no such gap: the session is standing, so it is named.
|
|
1632
|
+
|
|
1633
|
+
What that covers, in one rule:
|
|
1634
|
+
|
|
1635
|
+
- **Untagged** ``froid-loop-<id>`` sessions. The legacy pass runs with
|
|
1636
|
+
``require_tag=True`` (:func:`prunable_sessions`), so an untagged session there
|
|
1637
|
+
is skipped rather than claimed by a run dir that proves nothing in a shared
|
|
1638
|
+
registry.
|
|
1639
|
+
- **Ours, still standing.** Tagged this project's, and the sweep did not remove
|
|
1640
|
+
it — because it was live, because it exited mid-sweep, or because
|
|
1641
|
+
``kill_session`` (best-effort and silent by contract) did not land. All three
|
|
1642
|
+
leave the same fact behind: a session of ours in a registry ordinary attach
|
|
1643
|
+
and cleanup no longer address. Naming it needs no cause, which is why this
|
|
1644
|
+
also closes the failed-kill case ``cleanup --json``'s ``sessions.removed``
|
|
1645
|
+
documents as an *attempted* kill.
|
|
1646
|
+
- **A surviving control session.** The prune never touches a ctl-named
|
|
1647
|
+
session (:func:`is_ctl_session_name`), and its parked windows are not swept
|
|
1648
|
+
in a legacy registry either — the ctl-window scan runs against the primary
|
|
1649
|
+
backend only. The shape question is asked through *that registry's*
|
|
1650
|
+
:meth:`~.adapters.multiplexer.TerminalMultiplexer.session_name_key`, never a
|
|
1651
|
+
constant fold: on a case-folding store ``froid-loop-CTL-<hex>`` IS the
|
|
1652
|
+
control session and goes unnamed without it, while on an exact one it is a
|
|
1653
|
+
distinct session froid-loop cannot have minted — naming it there would send
|
|
1654
|
+
the operator after somebody else's.
|
|
1655
|
+
|
|
1656
|
+
Another project's tagged sessions never appear: the sweep skipping them is the
|
|
1657
|
+
correct outcome, not a remainder.
|
|
1658
|
+
|
|
1659
|
+
``announced`` is the one thing presence alone cannot judge: on a **dry run**
|
|
1660
|
+
nothing was killed, so every session the preview just announced as a would-kill
|
|
1661
|
+
is still standing and would be named here as if the sweep had declined it. The
|
|
1662
|
+
caller passes the run ids it printed — :func:`prune_sessions`' own return — and
|
|
1663
|
+
they are excluded.
|
|
1664
|
+
|
|
1665
|
+
**Excluded only where THIS registry's own pass could have announced it**, which
|
|
1666
|
+
is the tagged-ours arm and only it. :func:`prune_sessions` unions the ids of
|
|
1667
|
+
every pass, the *primary* registry's included, so the flat set says no more
|
|
1668
|
+
than "some registry would kill this id" — while a legacy pass runs with
|
|
1669
|
+
``require_tag=True`` and therefore cannot claim an untagged session at all.
|
|
1670
|
+
Applied to the untagged arm the set hid exactly the remainder this listing
|
|
1671
|
+
exists for: a dead ``froid-loop-X`` the primary pass plans to kill, an untagged
|
|
1672
|
+
``froid-loop-X`` over here that the real cleanup leaves and reports, and a
|
|
1673
|
+
preview of that same cleanup that does not mention it.
|
|
1674
|
+
|
|
1675
|
+
Inside the tagged arm the flat set is exact, so no per-registry plan has to be
|
|
1676
|
+
threaded down here. Liveness is read from ``run_dir_for(project, run_id)`` —
|
|
1677
|
+
one directory per (project, id), whatever registry the session sits in — so an
|
|
1678
|
+
id the primary pass judged dead the legacy pass judges dead too: if the same
|
|
1679
|
+
id is standing here under a tag proving ours, this pass announced it as well
|
|
1680
|
+
and the union merely collapsed the two.
|
|
1681
|
+
|
|
1682
|
+
Passed in rather than re-derived, and that is the whole point of the parameter.
|
|
1683
|
+
An earlier revision re-ran the partition here to rediscover the plan, which is
|
|
1684
|
+
a *second sample*: a tagged legacy run seen alive by the first (so printed as
|
|
1685
|
+
live, never announced) can exit before this call, land in the second sample's
|
|
1686
|
+
prunable arm, and be excluded from a listing it should have headed — a session
|
|
1687
|
+
dropped from the preview outright, not merely mentioned twice. Consuming what
|
|
1688
|
+
the preview actually printed cannot disagree with it.
|
|
1689
|
+
|
|
1690
|
+
On a real cleanup the caller passes nothing: there, a killed session is gone
|
|
1691
|
+
from the listing by presence, and one whose kill did not land must be named.
|
|
1692
|
+
|
|
1693
|
+
Deliberately its own listing rather than a fourth arm on
|
|
1694
|
+
:func:`prune_sessions`. That tuple is read by two frontends and projected into
|
|
1695
|
+
the schema-versioned ``cleanup --json`` document; widening it is a contract
|
|
1696
|
+
change and ~30 call sites, against one extra pair of psmux calls against a
|
|
1697
|
+
registry that answers "no server" instantly on any machine that never ran the
|
|
1698
|
+
pre-registry build.
|
|
1699
|
+
|
|
1700
|
+
Names, not run ids: the ctl session has no run id, and the operator is going to
|
|
1701
|
+
paste these into a ``psmux`` target — under the registry this maps them to,
|
|
1702
|
+
which is the other half of what makes them pasteable.
|
|
1703
|
+
"""
|
|
1704
|
+
grouped: dict[str, list[str]] = {}
|
|
1705
|
+
mine = accepted_tags(project)
|
|
1706
|
+
# Run ids, so names. `prune_sessions` unions its passes, so an id it reports
|
|
1707
|
+
# names at most one session anywhere — the same collapse that makes its own
|
|
1708
|
+
# "killed" count one per id.
|
|
1709
|
+
excluded = {session_name(run_id) for run_id in announced}
|
|
1710
|
+
for legacy in _legacy_registries():
|
|
1711
|
+
try:
|
|
1712
|
+
names = legacy.list_sessions()
|
|
1713
|
+
tags = legacy.session_options(PROJECT_OPTION) if names else {}
|
|
1714
|
+
except MultiplexerError:
|
|
1715
|
+
continue # observation degrades; the sweep's own report still stands
|
|
1716
|
+
here: list[str] = []
|
|
1717
|
+
for name in names:
|
|
1718
|
+
if is_ctl_session_name(legacy.session_name_key(name)):
|
|
1719
|
+
# A legacy registry holds the pre-#537 fixed name; the shape
|
|
1720
|
+
# predicate also names any per-registry-named stray. Asked
|
|
1721
|
+
# through THIS registry's own comparison key, never a constant
|
|
1722
|
+
# fold: whether `froid-loop-CTL-<hex>` denotes the control
|
|
1723
|
+
# session is the transport's answer to give.
|
|
1724
|
+
here.append(name)
|
|
1725
|
+
continue
|
|
1726
|
+
if _agent_run_id(name) is None:
|
|
1727
|
+
continue # not a froid-loop agent session at all
|
|
1728
|
+
tag = tags.get(name, "")
|
|
1729
|
+
if tag and tag not in mine:
|
|
1730
|
+
continue # another project's session
|
|
1731
|
+
if tag and name in excluded:
|
|
1732
|
+
continue # a would-kill of this registry's own pass (dry run)
|
|
1733
|
+
here.append(name)
|
|
1734
|
+
if here:
|
|
1735
|
+
# `registry_root()` is a diagnostic and never raises (seam contract).
|
|
1736
|
+
# Two admitted registries could in principle answer the same label —
|
|
1737
|
+
# a displaced root that spells the default is swept twice — so the
|
|
1738
|
+
# rows are merged rather than overwritten.
|
|
1739
|
+
label = legacy.registry_root() or DEFAULT_REGISTRY_LABEL
|
|
1740
|
+
grouped[label] = sorted(set(grouped.get(label, []) + here))
|
|
1741
|
+
return grouped
|
|
1742
|
+
|
|
1743
|
+
|
|
1744
|
+
def _legacy_registries() -> list[TerminalMultiplexer]:
|
|
1745
|
+
"""Backends bound to registries this project's sessions may predate, or []
|
|
1746
|
+
(see :meth:`~.multiplexer.TerminalMultiplexer.legacy_registries`, which owns
|
|
1747
|
+
the concept and every backend's answer).
|
|
1748
|
+
|
|
1749
|
+
Degrades to [] rather than raising: a backend that cannot even be selected
|
|
1750
|
+
has no legacy registry to offer, and a cleanup that already swept the primary
|
|
1751
|
+
registry must report that work rather than die on the migration pass."""
|
|
1752
|
+
try:
|
|
1753
|
+
return list(get_multiplexer().legacy_registries())
|
|
1754
|
+
except MultiplexerError:
|
|
1755
|
+
return []
|
|
1756
|
+
|
|
1757
|
+
|
|
1758
|
+
# The run dir of the OUTERMOST engine in this call stack (#319). A nested auto-sweep
|
|
1759
|
+
# runs synchronously in its parent's thread but mints its own run id and dir, so its
|
|
1760
|
+
# adapters would poll a control file no operator ever writes to: `froid-loop stop
|
|
1761
|
+
# <parent-id>` lodges in the parent's dir. This carries the owning run dir down to
|
|
1762
|
+
# them. A ContextVar, mirroring `engine._run_depth`, because the nesting it tracks is
|
|
1763
|
+
# same-thread by construction; set once by the outermost `Engine.run()` and reset by
|
|
1764
|
+
# token, so a later top-level run in the same process is never poisoned.
|
|
1765
|
+
_owner_run_dir: contextvars.ContextVar[Path | None] = contextvars.ContextVar(
|
|
1766
|
+
"froid_loop_owner_run_dir", default=None
|
|
1767
|
+
)
|
|
1768
|
+
|
|
1769
|
+
|
|
1770
|
+
def set_owner_run_dir(run_dir: Path) -> contextvars.Token[Path | None]:
|
|
1771
|
+
"""Claim ``run_dir`` as the owning run for this call stack. Returns the token the
|
|
1772
|
+
caller must hand to :func:`reset_owner_run_dir` from a ``finally``."""
|
|
1773
|
+
return _owner_run_dir.set(run_dir)
|
|
1774
|
+
|
|
1775
|
+
|
|
1776
|
+
def reset_owner_run_dir(token: contextvars.Token[Path | None]) -> None:
|
|
1777
|
+
"""Release the claim made by :func:`set_owner_run_dir`."""
|
|
1778
|
+
_owner_run_dir.reset(token)
|
|
1779
|
+
|
|
1780
|
+
|
|
1781
|
+
def owner_run_dir() -> Path | None:
|
|
1782
|
+
"""The outermost engine's run dir, or None outside any run — which is what a
|
|
1783
|
+
standalone adapter (tests, probes) reads, so callers fall back to their own."""
|
|
1784
|
+
return _owner_run_dir.get()
|
|
1785
|
+
|
|
1786
|
+
|
|
1787
|
+
def graceful_stop_requested(run_dir: Path) -> bool:
|
|
1788
|
+
"""True when *some* stop request is pending for this run — either mode. A bare
|
|
1789
|
+
existence read of the control file, never raising and deliberately never parsing.
|
|
1790
|
+
|
|
1791
|
+
Every consumer wants exactly that existence question, not the mode: the
|
|
1792
|
+
``stopping`` projection and the TUI badge (a run with a hard request lodged is
|
|
1793
|
+
stopping too), the ``--graceful`` idempotency check (a lodged hard request means
|
|
1794
|
+
a *stronger* stop already stands — "already-pending" is the right answer), the
|
|
1795
|
+
stories done-checkpoint skip, and auto-sweep suppression. Only ``status``'s
|
|
1796
|
+
``graceful_stop_pending`` field is mode-exact; it calls
|
|
1797
|
+
:func:`read_stop_request_mode` instead."""
|
|
1798
|
+
return (run_dir / STOP_REQUEST_FILE).is_file()
|
|
1799
|
+
|
|
1800
|
+
|
|
1801
|
+
def read_stop_request_mode(run_dir: Path) -> str | None:
|
|
1802
|
+
"""The mode of this run's pending stop request: ``"hard"``, ``"graceful"``, or
|
|
1803
|
+
``None`` when none is pending.
|
|
1804
|
+
|
|
1805
|
+
``None`` means *absent*, and only absent — it is returned for
|
|
1806
|
+
``FileNotFoundError`` alone. Everything else about a file that is *present*
|
|
1807
|
+
reads ``"graceful"``: a modeless body (every pre-#319 writer and test fixture
|
|
1808
|
+
wrote one — this is the back-compat pin), unparseable or non-object JSON, and a
|
|
1809
|
+
transient read failure such as the win32 sharing violation a concurrent
|
|
1810
|
+
``atomic_replace`` raises mid-write.
|
|
1811
|
+
|
|
1812
|
+
Leaning graceful on every ambiguity is load-bearing, not defensive habit. A
|
|
1813
|
+
misread graceful costs at most one more item before the run stops; a spurious
|
|
1814
|
+
``"hard"`` would abort a live session — so a torn read must never be able to
|
|
1815
|
+
produce one."""
|
|
1816
|
+
return _stop_request_mode_of(run_dir / STOP_REQUEST_FILE)
|
|
1817
|
+
|
|
1818
|
+
|
|
1819
|
+
def _stop_request_mode_of(path: Path) -> str | None:
|
|
1820
|
+
"""The parse half of :func:`read_stop_request_mode`, split out so
|
|
1821
|
+
:func:`consume_stop_request` can answer for the file it *took* rather than for
|
|
1822
|
+
whatever currently answers to the channel name."""
|
|
1823
|
+
try:
|
|
1824
|
+
raw = path.read_text(encoding="utf-8")
|
|
1825
|
+
except FileNotFoundError:
|
|
1826
|
+
return None
|
|
1827
|
+
except (OSError, ValueError):
|
|
1828
|
+
# present but unreadable this tick (sharing violation, undecodable bytes) —
|
|
1829
|
+
# answer for the file we know is there, never escalate on a failed read.
|
|
1830
|
+
return "graceful"
|
|
1831
|
+
try:
|
|
1832
|
+
body = json.loads(raw)
|
|
1833
|
+
except ValueError:
|
|
1834
|
+
return "graceful"
|
|
1835
|
+
if isinstance(body, dict) and body.get("mode") == "hard":
|
|
1836
|
+
return "hard"
|
|
1837
|
+
return "graceful"
|
|
1838
|
+
|
|
1839
|
+
|
|
1840
|
+
def _project_of_run_dir(run_dir: Path) -> Path:
|
|
1841
|
+
"""The project root a run directory hangs under, for confining writes into it.
|
|
1842
|
+
|
|
1843
|
+
Derived rather than passed because the stop-request channel is addressed by
|
|
1844
|
+
run directory alone: `stop_run` resolves a run reference and never holds the
|
|
1845
|
+
project separately. :func:`run_dir_for` is the only builder of these paths
|
|
1846
|
+
and spells them ``project / RUNS_DIR / run_id``, so the root sits exactly
|
|
1847
|
+
``len(RUNS_DIR.parts)`` levels above the run's own directory — the arithmetic
|
|
1848
|
+
tracks `RUNS_DIR` rather than hard-coding 2, so moving the runs tree moves
|
|
1849
|
+
this with it.
|
|
1850
|
+
|
|
1851
|
+
A path too shallow to have that ancestor is not one this module built.
|
|
1852
|
+
Refusing with :class:`UnconfinedWriteError` rather than letting `parents`
|
|
1853
|
+
raise `IndexError` is the load-bearing part: `stop_run` degrades on `OSError`
|
|
1854
|
+
so that a failed lodge still signals the run, and an `IndexError` there would
|
|
1855
|
+
abort the stop before it ever signalled."""
|
|
1856
|
+
depth = len(RUNS_DIR.parts)
|
|
1857
|
+
parents = run_dir.parents
|
|
1858
|
+
if len(parents) <= depth:
|
|
1859
|
+
raise UnconfinedWriteError(f"{run_dir} is not shaped like a run directory")
|
|
1860
|
+
return parents[depth]
|
|
1861
|
+
|
|
1862
|
+
|
|
1863
|
+
def _write_stop_request(run_dir: Path, mode: str) -> None:
|
|
1864
|
+
"""Lodge a stop request of ``mode`` on the control-file channel, written
|
|
1865
|
+
atomically so a concurrent engine read never sees a partial body.
|
|
1866
|
+
|
|
1867
|
+
The atomic replace *is* the supersede: writing ``"hard"`` over a pending
|
|
1868
|
+
``"graceful"`` escalates the request in one step, with no window in which
|
|
1869
|
+
nothing is pending for the engine to find.
|
|
1870
|
+
|
|
1871
|
+
That is the only direction this function arbitrates, and the only one it may:
|
|
1872
|
+
``stop_run`` shares it and its escalation must stay unconditional. The channel is
|
|
1873
|
+
otherwise last-writer-wins, so the *reverse* — a graceful write landing on a
|
|
1874
|
+
pending hard request and downgrading it — is refused by a different writer
|
|
1875
|
+
entirely: :func:`_create_stop_request`, which lodges the graceful mode with
|
|
1876
|
+
``O_CREAT | O_EXCL`` so "is one pending?" and "lodge mine" are a single atomic
|
|
1877
|
+
step. Splitting the two directions across two functions is what lets this one
|
|
1878
|
+
stay an unconditional replace.
|
|
1879
|
+
|
|
1880
|
+
Goes through :func:`platform_util.atomic_write_text` rather than a hand-rolled
|
|
1881
|
+
``tmp + atomic_replace``, for the reason ``operatoractions`` was migrated under
|
|
1882
|
+
#379: this is the one control file with genuinely *concurrent* writers — two
|
|
1883
|
+
``stop`` invocations against the same run, in either mode — and a fixed ``.tmp``
|
|
1884
|
+
sibling is exactly what two writers of the same key collide on. Interleaved,
|
|
1885
|
+
both stage over one name and the loser's ``os.replace`` raises
|
|
1886
|
+
``FileNotFoundError`` after the winner's consumed it; on the hard path that
|
|
1887
|
+
would abort ``stop_run`` *before* it ever signals. A ``mkstemp`` temp per writer
|
|
1888
|
+
removes the collision: the last replace wins and neither writer errors.
|
|
1889
|
+
|
|
1890
|
+
Refusing a link at the control file preserved what the bare ``os.replace``
|
|
1891
|
+
did — it never dereferenced this destination — and matches what the file is:
|
|
1892
|
+
machine-minted control state under a run dir a driven session can reach. The
|
|
1893
|
+
write is now confined to the project root (#593), because that refusal
|
|
1894
|
+
covered only the final component: every directory above it was still looked
|
|
1895
|
+
up by name, so a link planted at ``.froid-loop/`` — or at ``runs/``, or at the
|
|
1896
|
+
run's own directory — aimed both the temp and the publish wherever it
|
|
1897
|
+
pointed. The file still lands at ``mkstemp``'s ``0600`` instead of
|
|
1898
|
+
``0644 & ~umask``, since no-follow never inherited a mode either; nothing
|
|
1899
|
+
reads it cross-user.
|
|
1900
|
+
|
|
1901
|
+
No ``require_writable_target``: this is not an operator-curated file but a
|
|
1902
|
+
channel two ``stop`` invocations race on, and its whole contract above is
|
|
1903
|
+
that the stronger request always lands."""
|
|
1904
|
+
body = json.dumps({"requested_at": time.strftime("%Y-%m-%dT%H:%M:%S"), "mode": mode})
|
|
1905
|
+
atomic_write_text_confined(
|
|
1906
|
+
run_dir / STOP_REQUEST_FILE, body, confine_root=_project_of_run_dir(run_dir)
|
|
1907
|
+
)
|
|
1908
|
+
|
|
1909
|
+
|
|
1910
|
+
def _create_stop_request(run_dir: Path) -> bool:
|
|
1911
|
+
"""Lodge a *graceful* request only if none is pending; False when one already is.
|
|
1912
|
+
|
|
1913
|
+
``O_CREAT | O_EXCL`` is the arbitration. It makes "is a request pending?" and
|
|
1914
|
+
"lodge mine" one atomic step against the destination name, so a hard request
|
|
1915
|
+
landing at any instant either already exists — we refuse, leaving it standing —
|
|
1916
|
+
or replaces what we wrote, which is escalation, the direction
|
|
1917
|
+
:func:`_write_stop_request` owns. A re-read immediately before an unconditional
|
|
1918
|
+
replace could only ever *narrow* that window (~1.3ms on a journalling
|
|
1919
|
+
filesystem, where the fsync dominates); this closes it.
|
|
1920
|
+
|
|
1921
|
+
Graceful-ONLY by construction, and that is what makes the non-atomic body safe.
|
|
1922
|
+
The bytes are written *into* the created file rather than replaced in, so a
|
|
1923
|
+
concurrent reader can catch it empty — and :func:`read_stop_request_mode`
|
|
1924
|
+
answers ``"graceful"`` for a present-but-unparseable body, which is the very
|
|
1925
|
+
mode being written. The invariant that matters is untouched: a torn read must
|
|
1926
|
+
never produce ``"hard"``, so a hard writer must keep the atomic replace.
|
|
1927
|
+
|
|
1928
|
+
Refuses a planted symlink rather than following it — ``O_EXCL`` never
|
|
1929
|
+
dereferences — which is stricter than the ``follow_symlinks=False`` replace it
|
|
1930
|
+
replaces. That refusal covers only the FINAL component, though, so the create
|
|
1931
|
+
goes through :func:`platform_util.create_exclusive_confined` (#593): a link
|
|
1932
|
+
planted at ``.froid-loop/``, ``runs/`` or the run's own directory was still
|
|
1933
|
+
resolved by name and aimed the request outside the project, exactly the hole
|
|
1934
|
+
the confined :func:`_write_stop_request` next door already closed. The
|
|
1935
|
+
anchored create keeps the exclusive arbitration this function is built on;
|
|
1936
|
+
an unreachable parent raises ``UnconfinedWriteError``.
|
|
1937
|
+
|
|
1938
|
+
A failed write is deliberately NOT rolled back, and that is load-bearing rather
|
|
1939
|
+
than sloppy. ``unlink`` resolves a *name*, not the inode this call created, so a
|
|
1940
|
+
rollback here would delete whatever occupies the path at that moment — including
|
|
1941
|
+
a ``"hard"`` request a concurrent ``stop`` escalated onto it while this write was
|
|
1942
|
+
in flight. That is a ``hard -> absent`` drop, the one descent the mode lattice
|
|
1943
|
+
:func:`consume_stop_request` documents must never happen, and on native Windows
|
|
1944
|
+
it would silently withdraw the only channel that can stop the engine. Guarding it
|
|
1945
|
+
is not available: an "unlink only if still my inode" step does not exist as one
|
|
1946
|
+
atomic operation, and both check-then-unlink shapes measure *worse* than no guard
|
|
1947
|
+
at all — the check moves the decision earlier and the destructive act later by
|
|
1948
|
+
its own cost, shifting the window rather than narrowing it (inode compare 1.39x,
|
|
1949
|
+
mode compare 2.30x, over a rendezvous-synchronised escalation sweep on btrfs).
|
|
1950
|
+
|
|
1951
|
+
What a failed write leaves behind is a short or empty body, which
|
|
1952
|
+
:func:`read_stop_request_mode` reads as ``"graceful"`` — exactly the mode this
|
|
1953
|
+
function was asked to lodge, for the one caller (``stop --graceful``) that an
|
|
1954
|
+
operator drove. It does not wedge the channel: a later graceful ask answers
|
|
1955
|
+
"already-pending", a later *hard* stop supersedes it unconditionally, and
|
|
1956
|
+
``stop --cancel-graceful`` or ``resume`` withdraws it. Leaving a graceful request
|
|
1957
|
+
standing is the bounded direction this channel already leans on everywhere else."""
|
|
1958
|
+
body = json.dumps({"requested_at": time.strftime("%Y-%m-%dT%H:%M:%S"), "mode": "graceful"})
|
|
1959
|
+
path = run_dir / STOP_REQUEST_FILE
|
|
1960
|
+
try:
|
|
1961
|
+
fd = create_exclusive_confined(path, confine_root=_project_of_run_dir(run_dir))
|
|
1962
|
+
except FileExistsError:
|
|
1963
|
+
return False # a request is already pending — a planted link included
|
|
1964
|
+
with os.fdopen(fd, "w", encoding="utf-8") as fh:
|
|
1965
|
+
fh.write(body)
|
|
1966
|
+
return True
|
|
1967
|
+
|
|
1968
|
+
|
|
1969
|
+
def clear_graceful_stop(run_dir: Path) -> bool:
|
|
1970
|
+
"""Consume a pending stop request of *either* mode, returning True iff one was
|
|
1971
|
+
present and removed. Never raises: the engine calls this the moment it honors a
|
|
1972
|
+
request, a resume calls it to discard a stale one, and stop_run calls it on the
|
|
1973
|
+
paths where nothing is left alive to read what it lodged — a missing file
|
|
1974
|
+
(already consumed) or an unremovable one must not wedge any of them. Uses the
|
|
1975
|
+
same win32 sharing-violation retry the atomic write pairs with."""
|
|
1976
|
+
try:
|
|
1977
|
+
retrying_unlink(run_dir / STOP_REQUEST_FILE)
|
|
1978
|
+
except OSError:
|
|
1979
|
+
# FileNotFoundError (nothing pending) or a genuine removal failure — either
|
|
1980
|
+
# way nothing was discarded, and the caller must not see an exception.
|
|
1981
|
+
return False
|
|
1982
|
+
return True
|
|
1983
|
+
|
|
1984
|
+
|
|
1985
|
+
def consume_stop_request(run_dir: Path) -> str | None:
|
|
1986
|
+
"""Take the pending request off the channel and answer the mode of the very file
|
|
1987
|
+
removed, or ``None`` when none was pending.
|
|
1988
|
+
|
|
1989
|
+
The reader-side counterpart of :func:`_create_stop_request`'s
|
|
1990
|
+
``O_CREAT | O_EXCL``: the rename *is* the consume, so "what mode is pending?"
|
|
1991
|
+
and "take it" cannot disagree. A read followed by an unlink can, and the gap is
|
|
1992
|
+
not academic — a concurrent ``stop`` escalating to ``"hard"`` in between is
|
|
1993
|
+
deleted unread while the caller routes on the stale ``"graceful"`` it already
|
|
1994
|
+
holds.
|
|
1995
|
+
|
|
1996
|
+
Only that direction can lose anything, because the mode lattice is monotone:
|
|
1997
|
+
the graceful writer refuses to overwrite an existing request and the hard writer
|
|
1998
|
+
only ever writes ``"hard"``, so ``absent < graceful < hard`` until consumed. A
|
|
1999
|
+
stale ``"hard"`` read is therefore always still true; a stale ``"graceful"`` may
|
|
2000
|
+
not be.
|
|
2001
|
+
|
|
2002
|
+
Monotone requires that no writer *descends* either, which is why
|
|
2003
|
+
:func:`_create_stop_request` has no rollback on a failed write: an ``unlink``
|
|
2004
|
+
keyed on the path rather than the inode it created is a ``hard -> absent`` drop,
|
|
2005
|
+
and it would put a second way to lose a hard request in a *writer* — leaving the
|
|
2006
|
+
three read-then-unlink sites in ``engine.py`` that rely on this argument resting
|
|
2007
|
+
on something untrue.
|
|
2008
|
+
|
|
2009
|
+
Re-reading the mode immediately before the unlink does NOT fix this, and is a
|
|
2010
|
+
trap worth naming: measured over 4000 injected races it made the loss *more*
|
|
2011
|
+
likely, not less (164 -> 929 swallowed), because the extra read lengthens the
|
|
2012
|
+
interval an escalation has to land in. Narrowing a window is not closing it —
|
|
2013
|
+
only one atomic step is.
|
|
2014
|
+
|
|
2015
|
+
A hard request lodged *after* the take is a new request against a run already
|
|
2016
|
+
stopping. It stays at the canonical name for ``run()``'s finally to discard and
|
|
2017
|
+
journal as ``stop-request-discarded`` — a record, not a silent loss."""
|
|
2018
|
+
src = run_dir / STOP_REQUEST_FILE
|
|
2019
|
+
taken = run_dir / (STOP_REQUEST_FILE + ".consumed")
|
|
2020
|
+
try:
|
|
2021
|
+
atomic_replace(src, taken)
|
|
2022
|
+
except FileNotFoundError:
|
|
2023
|
+
return None
|
|
2024
|
+
except OSError:
|
|
2025
|
+
# Could not take it (read-only dir, a sharing violation past its retries).
|
|
2026
|
+
# Leave it on the channel and answer from the canonical name: the next
|
|
2027
|
+
# boundary re-asks, which is strictly better than losing the request.
|
|
2028
|
+
return read_stop_request_mode(run_dir)
|
|
2029
|
+
try:
|
|
2030
|
+
return _stop_request_mode_of(taken)
|
|
2031
|
+
finally:
|
|
2032
|
+
with contextlib.suppress(OSError):
|
|
2033
|
+
retrying_unlink(taken)
|
|
2034
|
+
|
|
2035
|
+
|
|
2036
|
+
def request_graceful_stop(run_dir: Path) -> str:
|
|
2037
|
+
"""Ask a live run to stop gracefully: finish the in-flight item (story ->
|
|
2038
|
+
dev/review/commit, or a sweep bundle through commit) cleanly, then finalize and
|
|
2039
|
+
stop — resumable, unlike the hard stop :func:`stop_run` delivers.
|
|
2040
|
+
|
|
2041
|
+
Delivery is the :data:`STOP_REQUEST_FILE` control file, written atomically by
|
|
2042
|
+
:func:`_write_stop_request` so a concurrent engine read never sees a partial file.
|
|
2043
|
+
Never signals the process and never writes ``journal.jsonl`` (engine-owned
|
|
2044
|
+
single-writer). Returns a status token for the caller to message on:
|
|
2045
|
+
|
|
2046
|
+
- ``"requested"`` — file written; a provably-live engine will honor it.
|
|
2047
|
+
- ``"already-pending"`` — a request was already on disk; left untouched so its
|
|
2048
|
+
original ``requested_at`` stands (idempotent — a second ask is a no-op). The
|
|
2049
|
+
token is mode-blind, and the pending request is not necessarily graceful: a
|
|
2050
|
+
*hard* one sits there at rest whenever a `stop` could not prove the engine
|
|
2051
|
+
dead, and one can also land while this call is in flight. A stronger stop
|
|
2052
|
+
stands either way and must not be downgraded — so callers message this token
|
|
2053
|
+
as a *stop request*, never as a graceful one (#319).
|
|
2054
|
+
- ``"requested-unverifiable"`` — file written, but engine liveness read
|
|
2055
|
+
``'unknown'`` (e.g. a win32 access-denied pid): the request stands and fires
|
|
2056
|
+
if an engine is in fact running; the caller warns that it can't confirm.
|
|
2057
|
+
|
|
2058
|
+
Raises :class:`GracefulStopError` when the run has already finished (nothing to
|
|
2059
|
+
stop) or its engine is provably dead (no consumer — ``resume`` is the tool).
|
|
2060
|
+
"""
|
|
2061
|
+
state = load_state(run_dir)
|
|
2062
|
+
if state.finished:
|
|
2063
|
+
raise GracefulStopError(f"run {run_dir.name} has already finished — nothing to stop")
|
|
2064
|
+
if graceful_stop_requested(run_dir):
|
|
2065
|
+
return "already-pending" # keep the original request's timestamp
|
|
2066
|
+
liveness = engine_liveness(run_dir)
|
|
2067
|
+
if liveness == "dead":
|
|
2068
|
+
raise GracefulStopError(
|
|
2069
|
+
f"run {run_dir.name} has no live engine — a graceful stop request would "
|
|
2070
|
+
f"never be consumed; use `froid-loop resume {run_dir.name}` to continue it"
|
|
2071
|
+
)
|
|
2072
|
+
# The write IS the check. The existence test at the top of this function is
|
|
2073
|
+
# separated from here by a pid-file read, a liveness probe and (formerly) a
|
|
2074
|
+
# mkstemp and an fsync — measured at ~1.3ms median on btrfs, wide enough for a
|
|
2075
|
+
# concurrent `stop` to lodge `"hard"` in between — and the channel is
|
|
2076
|
+
# last-writer-wins, so an unconditional replace here would silently *downgrade*
|
|
2077
|
+
# it and cost the abort the operator asked for. A re-read just before the replace
|
|
2078
|
+
# narrows that window; a create-if-absent removes it, because there is no longer
|
|
2079
|
+
# a gap between deciding and writing. "already-pending" is the same answer the
|
|
2080
|
+
# check at the top gives, and the right one either way: a lodged hard request is
|
|
2081
|
+
# a *stronger* stop already standing. Two concurrent *graceful* asks resolve the
|
|
2082
|
+
# same way, which is the documented idempotency — the first one's timestamp
|
|
2083
|
+
# stands. The escalation direction is untouched and stays unconditional.
|
|
2084
|
+
if not _create_stop_request(run_dir):
|
|
2085
|
+
return "already-pending"
|
|
2086
|
+
return "requested" if liveness == "alive" else "requested-unverifiable"
|
|
2087
|
+
|
|
2088
|
+
|
|
2089
|
+
def stop_run(run_dir: Path) -> bool:
|
|
2090
|
+
"""Stop a live run. Returns False if it was already finished.
|
|
2091
|
+
|
|
2092
|
+
The request is delivered two ways at once, and the engine wins whichever race
|
|
2093
|
+
it can: a ``mode: hard`` :data:`STOP_REQUEST_FILE` is lodged *first*, then the
|
|
2094
|
+
engine is signalled. SIGTERM is the POSIX fast path — the handler stops the run
|
|
2095
|
+
within the tick. The file is what makes the stop work where the signal cannot
|
|
2096
|
+
land: a native-Windows engine never receives an inter-process SIGTERM, so before
|
|
2097
|
+
#319 every Windows stop burned the full grace window into a blind force-kill.
|
|
2098
|
+
Now the engine reads the file at its next item boundary, or mid-session in the
|
|
2099
|
+
adapter wait loop, and performs its own teardown either way.
|
|
2100
|
+
|
|
2101
|
+
That ordering is the whole point: lodging before signalling means the engine can
|
|
2102
|
+
never exit the signal path having missed a request that was only written after.
|
|
2103
|
+
|
|
2104
|
+
Either way the engine stays the single writer of `stopped` (it marks the run,
|
|
2105
|
+
kills its in-flight agent window, and exits). Falls back to an external kill +
|
|
2106
|
+
mark when there is no live engine pid, it is a legacy run, or it does not exit
|
|
2107
|
+
in time. A wedged engine that ignores both channels past the grace window is
|
|
2108
|
+
force-killed — but only while we can still prove the pid is the same process we
|
|
2109
|
+
signalled (a pid-reuse guard); otherwise we raise StopRunError rather than risk
|
|
2110
|
+
killing an unrelated process.
|
|
2111
|
+
|
|
2112
|
+
The lodged file is consumed by whoever settles the run: the engine when it
|
|
2113
|
+
honors the request, or this function on the paths where nothing is left alive to
|
|
2114
|
+
read it. Both exceptions to that turn on the same question — did we ever *prove*
|
|
2115
|
+
the engine dead? Where we did not, the file stays lodged, because it is then the
|
|
2116
|
+
only channel that can still stop it: the StopRunError refusal below (we decline
|
|
2117
|
+
to force-kill an unverifiable pid), and the ``engine_may_live`` paths where the
|
|
2118
|
+
signal or the kill was refused outright rather than racing us to exit.
|
|
2119
|
+
|
|
2120
|
+
**Registry scope, stated because it is easy to read past.** The stop
|
|
2121
|
+
itself is registry-independent: both channels address the engine *process* —
|
|
2122
|
+
the request file lands in the run directory, the signal on the pid recorded
|
|
2123
|
+
there — and a run directory is per (project, run id), not per registry. So a
|
|
2124
|
+
pre-upgrade run living in a legacy psmux registry stops, and a still-live
|
|
2125
|
+
engine tears down its own window under the registry it was launched with.
|
|
2126
|
+
What is scoped is the backstop below: :func:`kill_session` addresses the
|
|
2127
|
+
registry THIS process exported, so an agent session an already-dead engine
|
|
2128
|
+
leaked in a legacy registry is not reached from here and the run is marked
|
|
2129
|
+
stopped with that session standing.
|
|
2130
|
+
|
|
2131
|
+
Deliberately not widened, and for the reason ``kill_session``'s own docstring
|
|
2132
|
+
gives: a by-name kill in a registry shared with other projects, without tag
|
|
2133
|
+
proof, could take a neighbour's same-named session — run ids are unique per
|
|
2134
|
+
project only. Both legacy registries are shared in exactly that sense. The
|
|
2135
|
+
displaced one is no exception: it is the *ambient* ``PSMUX_DATA_DIR`` this
|
|
2136
|
+
process found (:func:`~.adapters.psmux_backend.note_displaced_registry`), so
|
|
2137
|
+
a profile that exports one exports it into every project's shell and every
|
|
2138
|
+
one of them kept its pre-upgrade sessions there. That is why the legacy pass
|
|
2139
|
+
of :func:`prune_sessions` demands the tag in both, and it is the path that
|
|
2140
|
+
reaches such a session — ``froid-loop cleanup``, with
|
|
2141
|
+
:func:`legacy_registry_leftovers` naming whatever the tag rule leaves and the
|
|
2142
|
+
registry it is in.
|
|
2143
|
+
"""
|
|
2144
|
+
state = load_state(run_dir)
|
|
2145
|
+
if state.finished:
|
|
2146
|
+
return False
|
|
2147
|
+
|
|
2148
|
+
# Lodge the hard request before signalling. The atomic replace also supersedes a
|
|
2149
|
+
# pending *graceful* request in the same step: the operator escalated past it, and
|
|
2150
|
+
# a stronger request must never leave a window where nothing at all is pending.
|
|
2151
|
+
#
|
|
2152
|
+
# Degrade rather than abort when the lodge fails (read-only run dir, ENOSPC — and
|
|
2153
|
+
# the run's own session logs tee into this very directory, so a run can fill the
|
|
2154
|
+
# disk that then blocks stopping it). The doctrine's unit is the *repair*, not the
|
|
2155
|
+
# syscall: this stop is "delivered two ways at once" per the docstring above, so
|
|
2156
|
+
# failing the whole thing because one of two redundant channels failed would leave
|
|
2157
|
+
# a run alive that the pre-#319 signal path could still have killed. Keep the
|
|
2158
|
+
# signal, and stay loud where it actually matters — see the refusal branch below.
|
|
2159
|
+
try:
|
|
2160
|
+
_write_stop_request(run_dir, "hard")
|
|
2161
|
+
lodged = True
|
|
2162
|
+
except OSError:
|
|
2163
|
+
lodged = False
|
|
2164
|
+
|
|
2165
|
+
host = get_process_host()
|
|
2166
|
+
pid, identity = read_pid_identity(run_dir) # identity recorded at run start, not sampled now
|
|
2167
|
+
if pid is not None and identity is not None and not host.alive_and_ours(pid, identity):
|
|
2168
|
+
# the pid we recorded is already gone, or was reused by an unrelated
|
|
2169
|
+
# process before stop_run ran — never signal a stranger; mark stopped below.
|
|
2170
|
+
pid = None
|
|
2171
|
+
# Whether this call ever proved the engine dead. Only a confirmed death licenses
|
|
2172
|
+
# the fallback below to discard the request we lodged: while the engine may still
|
|
2173
|
+
# be running, that file is the one channel left that can stop it (on native
|
|
2174
|
+
# Windows it is the *only* one), so retracting it would throw away the very
|
|
2175
|
+
# repair #319 exists to deliver.
|
|
2176
|
+
engine_may_live = False
|
|
2177
|
+
if pid is not None:
|
|
2178
|
+
try:
|
|
2179
|
+
host.terminate(pid)
|
|
2180
|
+
except ProcessLookupError:
|
|
2181
|
+
pid = None # provably gone — the fallback's discard is correct
|
|
2182
|
+
except (PermissionError, OSError):
|
|
2183
|
+
# We could not signal it and it was `alive_and_ours` a moment ago, so it
|
|
2184
|
+
# may well still be running (an EPERM mismatch, or a win32 taskkill that
|
|
2185
|
+
# errored). Skip the wait — there is nothing to wait for — but keep the
|
|
2186
|
+
# request lodged so the engine can still stop itself off the file.
|
|
2187
|
+
engine_may_live = True
|
|
2188
|
+
pid = None
|
|
2189
|
+
if pid is not None:
|
|
2190
|
+
deadline = time.monotonic() + _STOP_WAIT_S
|
|
2191
|
+
while time.monotonic() < deadline:
|
|
2192
|
+
if not host.is_alive(pid):
|
|
2193
|
+
break # exited
|
|
2194
|
+
time.sleep(_STOP_POLL_S)
|
|
2195
|
+
if host.is_alive(pid):
|
|
2196
|
+
# still wedged past the grace window — escalate to a force-kill, but
|
|
2197
|
+
# only if this is provably the same process we signalled (never SIGKILL
|
|
2198
|
+
# a pid the kernel may have recycled to an unrelated process). For a
|
|
2199
|
+
# legacy pid file (no persisted identity) fall back to a stop-time
|
|
2200
|
+
# sample so a pre-upgrade run can still be force-killed — today's
|
|
2201
|
+
# behavior, carrying the same late-sample reuse window it always had.
|
|
2202
|
+
guard = identity if identity is not None else host.identity(pid)
|
|
2203
|
+
if guard is not None and host.identity(pid) == guard:
|
|
2204
|
+
try:
|
|
2205
|
+
host.force_kill(pid)
|
|
2206
|
+
except ProcessLookupError:
|
|
2207
|
+
pass # raced us to exit — that's the outcome we wanted
|
|
2208
|
+
except (PermissionError, OSError):
|
|
2209
|
+
# Unlike ESRCH above, this is the opposite news: the process is
|
|
2210
|
+
# there and we were refused. Keep the request lodged.
|
|
2211
|
+
engine_may_live = True
|
|
2212
|
+
else:
|
|
2213
|
+
# A kill that returned cleanly is not a death certificate — on
|
|
2214
|
+
# win32 `force_kill` shells `taskkill /F /T` with `check=False`,
|
|
2215
|
+
# so a refused kill raises nothing at all, and win32 is the
|
|
2216
|
+
# platform this whole channel exists for. Confirm rather than
|
|
2217
|
+
# infer, since the answer decides whether we discard the request.
|
|
2218
|
+
# Let it settle first: a delivered SIGKILL is immediate but the
|
|
2219
|
+
# pid can linger a moment before it is reaped, and reading that
|
|
2220
|
+
# as "still alive" would strand the file on the ordinary
|
|
2221
|
+
# wedged-engine path.
|
|
2222
|
+
confirm_deadline = time.monotonic() + _KILL_CONFIRM_S
|
|
2223
|
+
while host.is_alive(pid) and time.monotonic() < confirm_deadline:
|
|
2224
|
+
time.sleep(_STOP_POLL_S)
|
|
2225
|
+
engine_may_live = host.is_alive(pid)
|
|
2226
|
+
else:
|
|
2227
|
+
# Refusing to kill leaves the hard request lodged on purpose: if that
|
|
2228
|
+
# pid *is* still our engine, the file is the only channel left that
|
|
2229
|
+
# can stop it, and discarding it here would retract a request the
|
|
2230
|
+
# operator made while we decline to enforce it ourselves.
|
|
2231
|
+
#
|
|
2232
|
+
# That reasoning only holds while the lodge succeeded. If it did not,
|
|
2233
|
+
# nothing at all is pending and we are declining to force-kill on top
|
|
2234
|
+
# of that — the operator must be told, or they are left believing a
|
|
2235
|
+
# request is in flight that was never written.
|
|
2236
|
+
if lodged:
|
|
2237
|
+
raise StopRunError(
|
|
2238
|
+
f"run {run_dir.name}: engine pid {pid} honored neither the "
|
|
2239
|
+
"lodged stop request nor SIGTERM, and its identity can no "
|
|
2240
|
+
"longer be verified; refusing to force-kill a possibly-reused "
|
|
2241
|
+
"pid"
|
|
2242
|
+
)
|
|
2243
|
+
raise StopRunError(
|
|
2244
|
+
f"run {run_dir.name}: the stop request could not be written to "
|
|
2245
|
+
f"the run directory and engine pid {pid} did not honor SIGTERM; "
|
|
2246
|
+
"its identity can no longer be verified, so it will not be "
|
|
2247
|
+
"force-killed. No stop is pending — free space in the run "
|
|
2248
|
+
"directory and retry, or stop the process yourself"
|
|
2249
|
+
)
|
|
2250
|
+
# the engine clears its agent window itself, but kill the session as a backstop
|
|
2251
|
+
# in case it died before tearing it down. Ahead of everything below, because both
|
|
2252
|
+
# exits from here need it — an engine that honored the stop and died before
|
|
2253
|
+
# tearing its window down leaks the session just as surely as one we killed.
|
|
2254
|
+
# This is the one registry-scoped step of the stop (see the docstring): it
|
|
2255
|
+
# addresses the registry this process exported, and `cleanup`'s legacy pass is
|
|
2256
|
+
# what reaches a session left in an older one.
|
|
2257
|
+
kill_session(run_dir.name)
|
|
2258
|
+
state = load_state(run_dir)
|
|
2259
|
+
if state.stopped:
|
|
2260
|
+
# The engine honored the stop and is gone, and its own `run-stop` already
|
|
2261
|
+
# stands in the journal. Stamping `fallback=True` on top would describe an
|
|
2262
|
+
# engine that did its own teardown as one that had to be stopped from
|
|
2263
|
+
# outside. This check deliberately sits out here rather than inside the
|
|
2264
|
+
# `pid is not None` arm it used to live in: every path that clears `pid`
|
|
2265
|
+
# early — a pid that is no longer ours, a `terminate` that raced the exit
|
|
2266
|
+
# and got `ProcessLookupError`, a refusal that could not verify it — skipped
|
|
2267
|
+
# it and fell straight through to the append. The plainest case needs no race
|
|
2268
|
+
# at all: `stop` on a run a previous `stop` already stopped (`stopped` is set,
|
|
2269
|
+
# `finished` is not, so the guard at the top does not fire).
|
|
2270
|
+
#
|
|
2271
|
+
# It normally consumes the file on the way out; clear it belt-and-braces so a
|
|
2272
|
+
# run that is later resumed can never find our request still lodged and
|
|
2273
|
+
# re-stop at its first item. Safe on the `engine_may_live` paths too: a
|
|
2274
|
+
# written `stopped` *is* the engine reporting it honored the request, so
|
|
2275
|
+
# there is no live consumer left to strand.
|
|
2276
|
+
clear_graceful_stop(run_dir)
|
|
2277
|
+
return True
|
|
2278
|
+
|
|
2279
|
+
# Neither channel was delivered: nothing is lodged, and we never proved the engine
|
|
2280
|
+
# dead. This is the one outcome `stop` must not report as success — the operator is
|
|
2281
|
+
# left believing a request is in flight that was never written, while an engine we
|
|
2282
|
+
# could not signal keeps mutating the project. The pid-reuse guard above already
|
|
2283
|
+
# refuses for its own path; these are its siblings, and the only reason they stayed
|
|
2284
|
+
# quiet is that they clear `pid` and skip that block. Not a regression — on the
|
|
2285
|
+
# merge-base this was the state of *every* refused signal, because `stop_run` cleared
|
|
2286
|
+
# the request as its first statement — but the earlier decision to report success
|
|
2287
|
+
# rested on the request being retained, which is exactly what did not happen here.
|
|
2288
|
+
#
|
|
2289
|
+
# Placement is load-bearing, twice over. It sits *after* the session backstop
|
|
2290
|
+
# because refusing to report a stop is no reason to leak the window, and *after* the
|
|
2291
|
+
# `state.stopped` return because a run the engine already honored must not be
|
|
2292
|
+
# reported as a failure. Journal the attempt before raising: the `run-stop` append
|
|
2293
|
+
# below is skipped, and an unrecorded stop attempt is its own trap.
|
|
2294
|
+
if engine_may_live and not lodged:
|
|
2295
|
+
Journal(run_dir).append("run-stop-undelivered", pid=pid)
|
|
2296
|
+
raise StopRunError(
|
|
2297
|
+
f"run {run_dir.name}: the stop request could not be written to the run "
|
|
2298
|
+
"directory and the engine could not be proved dead, so no stop is pending. "
|
|
2299
|
+
"Its agent session was killed as a backstop. Free space in the run directory "
|
|
2300
|
+
"and retry, or stop the process yourself"
|
|
2301
|
+
)
|
|
2302
|
+
|
|
2303
|
+
# Fallback: no live engine (or it never confirmed). Mark it stopped here. Discard
|
|
2304
|
+
# the request first — nothing is left alive to consume it, and a file outliving
|
|
2305
|
+
# the run it asked to stop is a trap for the next resume.
|
|
2306
|
+
#
|
|
2307
|
+
# Unless we never actually proved that. Where the engine may still be running,
|
|
2308
|
+
# the request stays lodged and the stop is genuinely still in flight: the engine
|
|
2309
|
+
# honors the file at its next poll and writes `stopped` itself. Discarding it here
|
|
2310
|
+
# would leave a live engine with no channel left while we report the run stopped —
|
|
2311
|
+
# the stale-request trap above is the lesser of the two, and it only bites a run
|
|
2312
|
+
# that is later resumed, which this one cannot be until that engine exits.
|
|
2313
|
+
if not engine_may_live:
|
|
2314
|
+
clear_graceful_stop(run_dir)
|
|
2315
|
+
state.stopped = True
|
|
2316
|
+
save_state(run_dir, state)
|
|
2317
|
+
Journal(run_dir).append("run-stop", pid=pid, fallback=True)
|
|
2318
|
+
return True
|
|
2319
|
+
|
|
2320
|
+
|
|
2321
|
+
def live_session_may_be_ours(project: Path, run_id: str) -> bool:
|
|
2322
|
+
"""True when a live ``froid-loop-<id>`` session exists that this project cannot
|
|
2323
|
+
prove belongs to another one — the precondition of the removal guard below.
|
|
2324
|
+
|
|
2325
|
+
Ownership is read exactly as :func:`prunable_sessions` reads it. A tag outside
|
|
2326
|
+
:func:`accepted_tags` proves the session foreign, and a *tagged* session carries
|
|
2327
|
+
its own ownership proof, so it does not need this project's run dir at all:
|
|
2328
|
+
answering False there keeps the guard off a removal that provably strands
|
|
2329
|
+
nothing. Untagged, or tagged as ours, answers True — neither can be ruled out
|
|
2330
|
+
as depending on this run dir, and only the untagged case is load-bearing.
|
|
2331
|
+
|
|
2332
|
+
An observation, so it degrades rather than raising, and each read degrades in
|
|
2333
|
+
its own direction. A listing that cannot answer reads as "no session": that is
|
|
2334
|
+
already what the bundled backend returns for a missing multiplexer, a dead
|
|
2335
|
+
server or a failed query, and a guard that varied by backend would be worse
|
|
2336
|
+
than no guard. A tag that cannot be read is *not* proof the session is foreign,
|
|
2337
|
+
so it reads as untagged and the refusal stands — by then the listing has
|
|
2338
|
+
already established that a session is live.
|
|
2339
|
+
|
|
2340
|
+
Both reads are caught explicitly because the seam permits a raise: only
|
|
2341
|
+
`pipe_pane` and `kill_session` are contractually best-effort, so an
|
|
2342
|
+
out-of-tree backend raises :class:`MultiplexerError` here where the bundled
|
|
2343
|
+
one returns empty (docs/adapter-authoring-guide.md). The listing is checked
|
|
2344
|
+
first, so the tag query only runs on a name collision.
|
|
2345
|
+
|
|
2346
|
+
A stronger shape was built and withdrawn: a proof discipline (block unless
|
|
2347
|
+
the transport *proves* the session absent) fell to four consecutive reviews,
|
|
2348
|
+
each refuting its newest proof source — the transports genuinely offer none.
|
|
2349
|
+
psmux's registry is advisory and self-healing (its server re-creates a
|
|
2350
|
+
reaped port file on a 5 s tick, source-read at v3.3.8), a binary's PATH
|
|
2351
|
+
presence is per-process while the server is not, and the listing is
|
|
2352
|
+
load-sensitive; so a "proof of absence" either wedges every removal behind
|
|
2353
|
+
`--force` or quietly accepts a refutable proof. The degrade above is the
|
|
2354
|
+
guard's owner's documented trade, kept deliberately; the measured cost of
|
|
2355
|
+
the unobservable-multiplexer window is filed for that owner to revisit
|
|
2356
|
+
rather than overturned here.
|
|
2357
|
+
|
|
2358
|
+
Two registry-root-era additions on that unchanged contract:
|
|
2359
|
+
|
|
2360
|
+
**The control-alias discount.** An id whose session name is one of THE
|
|
2361
|
+
control session's own names — the fixed :data:`CTL_SESSION`, or this
|
|
2362
|
+
project's :func:`ctl_session_for` — answers False before any transport
|
|
2363
|
+
read: that session is the control plane's, never claimed through a run
|
|
2364
|
+
dir, so its liveness is not evidence about the run, and blocking removal
|
|
2365
|
+
on it wedged exactly the recovery (`froid-loop delete ctl`) the resume
|
|
2366
|
+
refusal points an operator at, for as long as the machine had a control
|
|
2367
|
+
session at all. This is the *instance* question, deliberately not
|
|
2368
|
+
:func:`run_id_aliases_control_session`'s shape question: on tmux a
|
|
2369
|
+
`main`-created run `ctl-<16 hex>` owns a genuine agent session distinct
|
|
2370
|
+
from the fixed name (measured: killing it exactly leaves `froid-loop-ctl`
|
|
2371
|
+
alive), and the shape discount destroyed its run dir without ever querying
|
|
2372
|
+
the mux. A namespace probe that cannot answer degrades to the fixed name
|
|
2373
|
+
alone — the *smaller* discount, which blocks more, the safe direction. A
|
|
2374
|
+
discount, not a proof source: it removes non-evidence, and never clears a
|
|
2375
|
+
removal on transport testimony.
|
|
2376
|
+
|
|
2377
|
+
**Transport-owned name comparison.** Every comparison goes through
|
|
2378
|
+
:meth:`session_name_key`, never a constant fold: psmux resolves names
|
|
2379
|
+
through a case-folding store, tmux is case-sensitive (both measured), and
|
|
2380
|
+
a constant ``.lower()`` discounted a persisted `CTL` run's genuinely live
|
|
2381
|
+
uppercase agent on tmux as "the control session" and deleted its run dir.
|
|
2382
|
+
On tmux the key is identity, so the listing and tag reads keep their
|
|
2383
|
+
historical exact comparison. Selecting that backend is itself part of the
|
|
2384
|
+
listing read — :func:`mux_sessions` selects inside the caught call — so it
|
|
2385
|
+
degrades the listing's way: a transport that cannot even be chosen (a
|
|
2386
|
+
persisted `[mux] backend` naming a backend no longer registered) reports
|
|
2387
|
+
no live session rather than aborting every removal path."""
|
|
2388
|
+
try:
|
|
2389
|
+
mux = get_multiplexer()
|
|
2390
|
+
except MultiplexerError:
|
|
2391
|
+
return False
|
|
2392
|
+
key = mux.session_name_key
|
|
2393
|
+
name = session_name(run_id)
|
|
2394
|
+
control = {CTL_SESSION}
|
|
2395
|
+
try:
|
|
2396
|
+
control.add(ctl_session_for(project, mux))
|
|
2397
|
+
except MultiplexerError:
|
|
2398
|
+
pass # namespace unanswerable: only the fixed name is knowable
|
|
2399
|
+
if key(name) in {key(c) for c in control}:
|
|
2400
|
+
return False
|
|
2401
|
+
try:
|
|
2402
|
+
if key(name) not in {key(s) for s in mux_sessions()}:
|
|
2403
|
+
return False
|
|
2404
|
+
except MultiplexerError:
|
|
2405
|
+
return False
|
|
2406
|
+
try:
|
|
2407
|
+
tags = session_project_tags()
|
|
2408
|
+
except MultiplexerError:
|
|
2409
|
+
tags = {} # unread is not proof of foreign
|
|
2410
|
+
tag = next((v for s, v in tags.items() if key(s) == key(name)), "")
|
|
2411
|
+
return not tag or tag in accepted_tags(project)
|
|
2412
|
+
|
|
2413
|
+
|
|
2414
|
+
def _refuse_live_session(project: Path, run_id: str, verb: str) -> None:
|
|
2415
|
+
"""Backstop for #419: refuse to remove a run dir out from under a live session.
|
|
2416
|
+
|
|
2417
|
+
Every caller's live guard is keyed on *engine pid* liveness, so an orphan —
|
|
2418
|
+
engine dead, agent session still alive in the multiplexer — passes all of them.
|
|
2419
|
+
That is the one state where the run dir is load-bearing: for an untagged
|
|
2420
|
+
session it is the only ownership proof :func:`prunable_sessions` can read, so
|
|
2421
|
+
removing it leaks the session (and its server) for the life of the machine.
|
|
2422
|
+
Refusing is a repair-path write failing loudly, per the module doctrine.
|
|
2423
|
+
|
|
2424
|
+
Scoped to what it can justify: a session this project can prove is another
|
|
2425
|
+
one's does not block anything (see :func:`live_session_may_be_ours`). Refusing
|
|
2426
|
+
there would strand nothing and wedge every removal path — including `clean`,
|
|
2427
|
+
which has no override — for as long as the other project's run lives.
|
|
2428
|
+
|
|
2429
|
+
Never a kill from here: a session name carries no project, so killing
|
|
2430
|
+
`froid-loop-<id>` by name would tear down another project's live run whenever the
|
|
2431
|
+
two share a run id (reachable — `--run-id` is caller-supplied).
|
|
2432
|
+
|
|
2433
|
+
The message names `froid-loop cleanup` as the remedy but does not call it sound.
|
|
2434
|
+
`prune_sessions` proves ownership from the tag when there is one and falls back
|
|
2435
|
+
to *this same run dir* when there is not — the weak proof this guard exists to
|
|
2436
|
+
protect, so on the untagged case it can prune another project's session on a
|
|
2437
|
+
shared run id (#419's second edge, pinned by
|
|
2438
|
+
`test_prunable_sessions_claims_an_untagged_session_on_a_run_id_collision`).
|
|
2439
|
+
Hence the message asks the operator to confirm first: nothing available here can
|
|
2440
|
+
prove the session ours, and minting a proof that outlives the run dir is #419
|
|
2441
|
+
direction (2), not this guard."""
|
|
2442
|
+
if live_session_may_be_ours(project, run_id):
|
|
2443
|
+
raise LiveSessionError(
|
|
2444
|
+
f"run {run_id}: refusing to {verb} its directory while its agent session is "
|
|
2445
|
+
f"still live — for an untagged session this directory is the only ownership "
|
|
2446
|
+
f"proof a later prune has. Clear the session with `froid-loop cleanup` first, "
|
|
2447
|
+
f"having confirmed it is this project's (`froid-loop attach {run_id}`): an "
|
|
2448
|
+
f"untagged session is proven ours by this same directory, so a run id shared "
|
|
2449
|
+
f"with another project would prune theirs"
|
|
2450
|
+
)
|
|
2451
|
+
|
|
2452
|
+
|
|
2453
|
+
def _discard_state_dir(project: Path, run_id: str) -> None:
|
|
2454
|
+
"""Remove the run's out-of-tree control-plane counterpart, best-effort.
|
|
2455
|
+
|
|
2456
|
+
The events channel (#494) lives outside the project tree, so removing a run
|
|
2457
|
+
dir no longer removes everything the run owns: without this every
|
|
2458
|
+
delete/archive would leak ``<state root>/<project>/<run-id>/`` forever. It
|
|
2459
|
+
lives here rather than in the CLI so every caller inherits it — `delete`,
|
|
2460
|
+
`archive`, `clean`, the TUI's removal actions and the engine's own
|
|
2461
|
+
finish-time reclamation alike.
|
|
2462
|
+
|
|
2463
|
+
A **never-raise tail**, per the teardown doctrine (#139): the run dir is
|
|
2464
|
+
already gone by the time this runs, and failing the operator's delete over an
|
|
2465
|
+
unreachable state root would report a removal that in fact happened. Every
|
|
2466
|
+
catchable outcome is "the counterpart could not even be named" —
|
|
2467
|
+
:class:`StateRootError` for an environment with no derivable root, and
|
|
2468
|
+
``OSError``/``RuntimeError`` for a project path the OS cannot canonicalize
|
|
2469
|
+
(:func:`project_tag` resolves before digesting). ``RuntimeError`` is not
|
|
2470
|
+
optional there: below 3.13 ``Path.resolve`` reports a symlink loop that way
|
|
2471
|
+
rather than as ``OSError`` (measured — 3.11 and 3.12 raise, 3.13 and 3.14
|
|
2472
|
+
return the unresolved path), so on two supported interpreters an ``OSError``
|
|
2473
|
+
-only guard lets a loop escape and breaks the promise in this paragraph.
|
|
2474
|
+
Removal failures are absorbed by ``ignore_errors``. Either way the orphan
|
|
2475
|
+
sweep in :func:`reconcile_orphan_state_dirs` is the backstop.
|
|
2476
|
+
|
|
2477
|
+
Deliberately not called by :func:`trim_run_dir`: a trimmed run is still live
|
|
2478
|
+
on disk and resumable, and its control plane must outlive the scaffolding.
|
|
2479
|
+
"""
|
|
2480
|
+
try:
|
|
2481
|
+
target = state_dir_for(project, run_id)
|
|
2482
|
+
except (StateRootError, OSError, RuntimeError):
|
|
2483
|
+
return
|
|
2484
|
+
shutil.rmtree(target, ignore_errors=True)
|
|
2485
|
+
|
|
2486
|
+
|
|
2487
|
+
def _refuse_uncontained_run_dir(project: Path, run_dir: Path, action: str) -> None:
|
|
2488
|
+
"""Refuse to remove anything but a direct child of ``project``'s runs dir.
|
|
2489
|
+
|
|
2490
|
+
The containment half of #480, and deliberately independent of how the ref was
|
|
2491
|
+
spelled: :func:`_is_path_escape` gates the *string* an operator typed, this
|
|
2492
|
+
gates the *path* the two destructive writes are about to hand `shutil.rmtree`.
|
|
2493
|
+
Both are wanted. `delete_run` and `archive_run` are module-public and take a
|
|
2494
|
+
`run_dir` outright, so a caller that composed one by some route other than
|
|
2495
|
+
:func:`resolve_run_dir` — the TUI's selection, a record read back from disk, a
|
|
2496
|
+
call site not yet written — never passes the ref guard at all.
|
|
2497
|
+
|
|
2498
|
+
``run_dir_for`` is the sole builder of these paths, so recomposing one from
|
|
2499
|
+
the basename and comparing is exactly the "is a direct child" question: the
|
|
2500
|
+
runs root itself, a nested grandchild, and anything outside the project all
|
|
2501
|
+
differ from what it returns. Comparing against the rebuild rather than
|
|
2502
|
+
walking `parents` keeps this tracking `RUNS_DIR` the way
|
|
2503
|
+
:func:`_project_of_run_dir` does. The rebuild has one blind spot the name
|
|
2504
|
+
check closes: ``.name`` of ``runs / ".."`` is ``".."`` and the rebuild
|
|
2505
|
+
reproduces it verbatim, so the lexical equality holds while `rmtree` would
|
|
2506
|
+
resolve it to ``.froid-loop`` itself. pathlib drops ``"."`` at parse so only
|
|
2507
|
+
the ``".."`` spelling survives to here; ``"."`` is refused anyway rather than
|
|
2508
|
+
reasoned about.
|
|
2509
|
+
|
|
2510
|
+
The link walk below the equality check refuses a REDIRECTED spelling of a
|
|
2511
|
+
contained path: with ``.froid-loop``, ``runs`` or the run dir itself replaced
|
|
2512
|
+
by a symlink (or, on Windows, an unelevated ``mklink /J`` junction — why this
|
|
2513
|
+
is :func:`is_link_like` and not ``is_symlink``), the rebuild is lexically
|
|
2514
|
+
identical while `rmtree` follows the redirect and removes a tree outside the
|
|
2515
|
+
project. A planted redirect is this module's live threat class (see the #591
|
|
2516
|
+
notes in :func:`archive_run`). The walk stops short of ``project`` — the
|
|
2517
|
+
operator's own argument, and a project addressed through a symlinked home is
|
|
2518
|
+
legitimate — and covers only the orchestrator-owned levels under it. It is
|
|
2519
|
+
check-then-act, not fd-anchored like `journal.py`'s writes: `resolve()` is
|
|
2520
|
+
banned here (it can raise on a WSL-UNC host — `tests/conftest.py`'s
|
|
2521
|
+
``refuse_to_resolve``), `tarfile` cannot take a dir fd at all, and the racer
|
|
2522
|
+
that could re-plant between check and rmtree is a live session, which the
|
|
2523
|
+
guard below this one refuses anyway.
|
|
2524
|
+
|
|
2525
|
+
Raises rather than degrading — observation may degrade, a repair write must
|
|
2526
|
+
not: there is no partial `rmtree` to fall back to, and declining quietly would
|
|
2527
|
+
report a removal that never happened. :class:`UnconfinedWriteError` is the
|
|
2528
|
+
shape-refusal this module already raises for the same class of mistake (see
|
|
2529
|
+
:func:`_project_of_run_dir`), and being an ``OSError`` it lands in the
|
|
2530
|
+
handling callers already have for a removal that failed."""
|
|
2531
|
+
if run_dir.name in (".", "..") or run_dir_for(project, run_dir.name) != run_dir:
|
|
2532
|
+
raise UnconfinedWriteError(
|
|
2533
|
+
f"refusing to {action} {run_dir}: not a run directory under {project / RUNS_DIR}"
|
|
2534
|
+
)
|
|
2535
|
+
node = run_dir
|
|
2536
|
+
while node != project:
|
|
2537
|
+
if is_link_like(node):
|
|
2538
|
+
raise UnconfinedWriteError(
|
|
2539
|
+
f"refusing to {action} {run_dir}: {node} is a symlink or junction"
|
|
2540
|
+
)
|
|
2541
|
+
parent = node.parent
|
|
2542
|
+
if parent == node: # anchored: never walk past the filesystem root
|
|
2543
|
+
break
|
|
2544
|
+
node = parent
|
|
2545
|
+
|
|
2546
|
+
|
|
2547
|
+
def delete_run(project: Path, run_dir: Path, *, force: bool = False) -> None:
|
|
2548
|
+
"""Permanently remove a run directory. Callers enforce the engine-liveness
|
|
2549
|
+
guard; the session guard is enforced here (see :func:`_refuse_live_session`),
|
|
2550
|
+
which raises :class:`LiveSessionError` instead of removing.
|
|
2551
|
+
|
|
2552
|
+
``force`` is the operator's explicit override and skips that guard, accepting
|
|
2553
|
+
the leak on their own say-so. It deliberately does not kill the session
|
|
2554
|
+
instead — that would be unscoped, and this project cannot prove the session is
|
|
2555
|
+
its own (which is the whole defect). Trading a possible leak of our own session
|
|
2556
|
+
for a possible kill of someone else's is the wrong direction for an override.
|
|
2557
|
+
|
|
2558
|
+
The containment guard runs first and is NOT under ``force``: an override is
|
|
2559
|
+
the operator accepting a leaked session, never a licence to rmtree a path
|
|
2560
|
+
outside the runs dir."""
|
|
2561
|
+
_refuse_uncontained_run_dir(project, run_dir, "delete")
|
|
2562
|
+
if not force:
|
|
2563
|
+
_refuse_live_session(project, run_dir.name, "delete")
|
|
2564
|
+
shutil.rmtree(run_dir)
|
|
2565
|
+
# after the run dir, never before: a raise above leaves the run whole, and a
|
|
2566
|
+
# whole run keeps its control plane (see _discard_state_dir).
|
|
2567
|
+
_discard_state_dir(project, run_dir.name)
|
|
2568
|
+
|
|
2569
|
+
|
|
2570
|
+
def archive_run(project: Path, run_dir: Path, *, force: bool = False) -> Path:
|
|
2571
|
+
"""Compress a run dir into .froid-loop/archive/<id>.tar.gz and remove the
|
|
2572
|
+
original. The tarball is written to a temp path then atomically replaced into
|
|
2573
|
+
place so a partial archive never appears. Callers enforce the engine-liveness
|
|
2574
|
+
guard; the session guard is enforced here (see :func:`_refuse_live_session`,
|
|
2575
|
+
and :func:`delete_run` for ``force``) and runs before the tarball is written,
|
|
2576
|
+
so a refusal leaves nothing behind.
|
|
2577
|
+
|
|
2578
|
+
The tarball holds the run dir only, so since #494 an archive no longer carries
|
|
2579
|
+
the run's ``events/``: the channel moved out of the tree, and its files are
|
|
2580
|
+
transient completion signals the watcher has already consumed — the recorded
|
|
2581
|
+
decision accepts losing them from the archive. Everything an archive is read
|
|
2582
|
+
for later (state, journal, tasks, logs) is in the run dir and unaffected.
|
|
2583
|
+
|
|
2584
|
+
Containment (see :func:`_refuse_uncontained_run_dir`) is checked ahead of both,
|
|
2585
|
+
for the reason the session guard runs early: a refusal must leave no archive
|
|
2586
|
+
directory and no tarball behind."""
|
|
2587
|
+
_refuse_uncontained_run_dir(project, run_dir, "archive")
|
|
2588
|
+
if not force:
|
|
2589
|
+
_refuse_live_session(project, run_dir.name, "archive")
|
|
2590
|
+
archive_dir = project / ARCHIVE_DIR
|
|
2591
|
+
archive_dir.mkdir(parents=True, exist_ok=True)
|
|
2592
|
+
dest = archive_dir / f"{run_dir.name}.tar.gz"
|
|
2593
|
+
# #363: the guard, not a helper — the path is handed to `tarfile.open`, so there
|
|
2594
|
+
# is no payload for `atomic_write_*` to take. Nothing gitignores this directory:
|
|
2595
|
+
# init writes `.froid-loop/runs/`, `.froid-loop/cache/`, `.froid-loop/policy.toml`
|
|
2596
|
+
# and `_froid/render/`, and `archive/` matches none of them. So a stranded temp
|
|
2597
|
+
# here is an untracked file holding `worktree_clean` False until a human removes
|
|
2598
|
+
# it — the same exposure `decisions._write_store`, `policy.write_mux_backend` and
|
|
2599
|
+
# `tui.settings.PolicyDoc.save` had. (Not the sweep's two `decisions.json`
|
|
2600
|
+
# writes, which look like the same fix but write under the ignored run dir.)
|
|
2601
|
+
#
|
|
2602
|
+
# #591: staged through `_mkstemp_beside` — the atomic writers' own exclusive
|
|
2603
|
+
# `0600` create (binary-mode on win32), under a fresh unpredictable name per
|
|
2604
|
+
# attempt. A fixed name made a temp stranded by a kill, or planted at the
|
|
2605
|
+
# guessable spelling, deny every later attempt as `FileExistsError`; the
|
|
2606
|
+
# truncate-and-reuse it replaced followed a planted symlink instead. mkstemp's
|
|
2607
|
+
# exclusivity still never opens a name something else holds, and the name being
|
|
2608
|
+
# this process's own mint is what licenses the cleanup unlink below. It sits
|
|
2609
|
+
# outside the `try` on purpose: a create that fails has staged nothing to
|
|
2610
|
+
# clean up.
|
|
2611
|
+
fd, tmp_name = _mkstemp_beside(dest)
|
|
2612
|
+
tmp = Path(tmp_name)
|
|
2613
|
+
try:
|
|
2614
|
+
with os.fdopen(fd, "wb") as raw:
|
|
2615
|
+
with tarfile.open(fileobj=raw, mode="w:gz") as tar:
|
|
2616
|
+
tar.add(run_dir, arcname=run_dir.name)
|
|
2617
|
+
# Flushed and fsynced before the publish, and unlike the rest of this
|
|
2618
|
+
# family that is not about staleness but about data loss: `shutil.rmtree`
|
|
2619
|
+
# below removes the only other copy of the run, so a crash with the
|
|
2620
|
+
# tarball still in page cache destroys it outright. Ordered inside the
|
|
2621
|
+
# fdopen context so the gzip trailer `tar.close()` just wrote is included.
|
|
2622
|
+
raw.flush()
|
|
2623
|
+
os.fsync(raw.fileno())
|
|
2624
|
+
atomic_replace(tmp, dest)
|
|
2625
|
+
except BaseException:
|
|
2626
|
+
with contextlib.suppress(OSError):
|
|
2627
|
+
tmp.unlink(missing_ok=True) # provably ours: mkstemp minted the name
|
|
2628
|
+
raise
|
|
2629
|
+
shutil.rmtree(run_dir)
|
|
2630
|
+
_discard_state_dir(project, run_dir.name) # same tail as delete_run
|
|
2631
|
+
return dest
|
|
2632
|
+
|
|
2633
|
+
|
|
2634
|
+
# ------------------------------------------------------- reclaim / retention
|
|
2635
|
+
|
|
2636
|
+
# Heavy per-run scaffolding trimmed from a concluded run dir while the
|
|
2637
|
+
# TUI-visible core (state.json, journal.jsonl, logs/, ATTENTION) is preserved,
|
|
2638
|
+
# so the run still lists and renders in the dashboard. "worktrees" mirrors
|
|
2639
|
+
# workspace.WORKTREE_DIRNAME; kept literal here to avoid an import cycle
|
|
2640
|
+
# (workspace imports nothing from runs, but runs stays leaf-light on purpose).
|
|
2641
|
+
#
|
|
2642
|
+
# VERIFY_DIR is the retained verifier stdout/stderr store. It qualifies as heavy
|
|
2643
|
+
# on the same measure as a worktree checkout: `[verify] stream_capture_kb`
|
|
2644
|
+
# defaults to 256 KiB per stream, so a run accumulates up to 512 KiB per verify
|
|
2645
|
+
# command per attempt, and nothing else ever reclaims it. Its journal records
|
|
2646
|
+
# survive the trim and keep naming the files (`stdout_path`/`stderr_path`), which
|
|
2647
|
+
# is the same bargain `worktrees` already makes — a trimmed run is a run you can
|
|
2648
|
+
# still see and resume, not one you can still re-read every artifact of. Imported
|
|
2649
|
+
# from the writer rather than re-spelled, so the reclaim cannot drift from the
|
|
2650
|
+
# directory `Journal.write_verify_stream` actually creates.
|
|
2651
|
+
_HEAVY_RUN_ENTRIES = ("worktrees", VERIFY_DIR)
|
|
2652
|
+
|
|
2653
|
+
|
|
2654
|
+
def heavy_run_entries(run_dir: Path) -> list[Path]:
|
|
2655
|
+
"""The paths :func:`trim_run_dir` would remove from ``run_dir``.
|
|
2656
|
+
|
|
2657
|
+
Exists so a caller sizing the reclaim measures exactly what the trim takes.
|
|
2658
|
+
`clean` sums these before mutating (its estimate has to hold under
|
|
2659
|
+
--dry-run); reading the tuple through this function is what keeps that sum
|
|
2660
|
+
from silently going stale the next time an entry is added to it."""
|
|
2661
|
+
return [run_dir / name for name in _HEAVY_RUN_ENTRIES]
|
|
2662
|
+
|
|
2663
|
+
|
|
2664
|
+
def _state_or_none(run_dir: Path):
|
|
2665
|
+
"""Parsed run state, or None when it cannot be read — never classify (and so
|
|
2666
|
+
never reclaim) what you cannot positively read."""
|
|
2667
|
+
try:
|
|
2668
|
+
return load_state(run_dir)
|
|
2669
|
+
except Exception: # unreadable/corrupt state ⇒ leave it alone
|
|
2670
|
+
return None
|
|
2671
|
+
|
|
2672
|
+
|
|
2673
|
+
def is_finished(run_dir: Path) -> bool:
|
|
2674
|
+
"""A finished, no-longer-live run. `resume` refuses these (cli checks
|
|
2675
|
+
state.finished), so tearing down their worktrees can never strand a resume —
|
|
2676
|
+
the safe predicate for the *automatic* reconcile paths."""
|
|
2677
|
+
if engine_alive(run_dir):
|
|
2678
|
+
return False
|
|
2679
|
+
state = _state_or_none(run_dir)
|
|
2680
|
+
return bool(state and state.finished)
|
|
2681
|
+
|
|
2682
|
+
|
|
2683
|
+
def reclaimable(run_dir: Path) -> bool:
|
|
2684
|
+
"""A terminal run (finished or stopped) with no live engine — eligible for
|
|
2685
|
+
the *explicit* `clean` command. A stopped run is technically resumable, so
|
|
2686
|
+
reclaiming its worktree ends that; `clean` is an opt-in reclaim (guarded by
|
|
2687
|
+
--keep / --dry-run). Paused, interrupted (crashed) and running/unknown-host
|
|
2688
|
+
runs are never reclaimed: paused/interrupted are actively resumable, and a
|
|
2689
|
+
missing pid could mean a foreign-host run, so we require positive local
|
|
2690
|
+
termination evidence (finished or stopped)."""
|
|
2691
|
+
if engine_alive(run_dir):
|
|
2692
|
+
return False
|
|
2693
|
+
state = _state_or_none(run_dir)
|
|
2694
|
+
return bool(state and (state.finished or state.stopped))
|
|
2695
|
+
|
|
2696
|
+
|
|
2697
|
+
def reconcile_orphan_worktrees(repo: Path, run_dir: Path, *, dry_run: bool = False) -> list[Path]:
|
|
2698
|
+
"""Force-remove every git worktree whose path lies under ``run_dir``, then
|
|
2699
|
+
prune git's admin entries. Reconciles from ``git worktree list`` (on-disk
|
|
2700
|
+
truth), NOT from policy — orphans created under a previous isolation=worktree
|
|
2701
|
+
config persist after a switch back to isolation=none. Returns the worktree
|
|
2702
|
+
paths handled (or that would be, under dry_run). Callers gate on
|
|
2703
|
+
``reclaimable``; the main checkout is never under a run dir, so it is safe."""
|
|
2704
|
+
run_res = run_dir.resolve()
|
|
2705
|
+
try:
|
|
2706
|
+
worktrees = verify.worktree_list(repo)
|
|
2707
|
+
except verify.GitError:
|
|
2708
|
+
return []
|
|
2709
|
+
handled: list[Path] = []
|
|
2710
|
+
for wt in worktrees:
|
|
2711
|
+
try:
|
|
2712
|
+
wt.resolve().relative_to(run_res)
|
|
2713
|
+
except (ValueError, OSError):
|
|
2714
|
+
continue # not this run's worktree (incl. the main checkout)
|
|
2715
|
+
handled.append(wt)
|
|
2716
|
+
if not dry_run:
|
|
2717
|
+
try:
|
|
2718
|
+
verify.worktree_remove(repo, wt, force=True)
|
|
2719
|
+
except verify.GitError:
|
|
2720
|
+
shutil.rmtree(wt, ignore_errors=True)
|
|
2721
|
+
if handled and not dry_run:
|
|
2722
|
+
verify.worktree_prune(repo)
|
|
2723
|
+
return handled
|
|
2724
|
+
|
|
2725
|
+
|
|
2726
|
+
def reconcile_stale_worktrees(repo: Path, project: Path, *, dry_run: bool = False) -> list[Path]:
|
|
2727
|
+
"""Safety net for the automatic paths (run/sweep start): tear down worktrees
|
|
2728
|
+
left behind by a *finished* run whose clean-finish GC didn't complete (e.g. a
|
|
2729
|
+
crash between merge and teardown). Deliberately finished-ONLY — a stopped run
|
|
2730
|
+
is still resumable, so its worktree is left for `resume`/`clean` to handle and
|
|
2731
|
+
never stranded out from under the operator."""
|
|
2732
|
+
handled: list[Path] = []
|
|
2733
|
+
for run_dir in list_run_dirs(project):
|
|
2734
|
+
if not is_finished(run_dir):
|
|
2735
|
+
continue
|
|
2736
|
+
handled += reconcile_orphan_worktrees(repo, run_dir, dry_run=dry_run)
|
|
2737
|
+
return handled
|
|
2738
|
+
|
|
2739
|
+
|
|
2740
|
+
def _run_dir_names(project: Path) -> set[str] | None:
|
|
2741
|
+
"""Every *directory name* under the runs dir, or ``None`` when that listing
|
|
2742
|
+
could not be taken.
|
|
2743
|
+
|
|
2744
|
+
Deliberately not :func:`list_run_dirs`, which is ``state.json``-gated: this
|
|
2745
|
+
answers "does a run dir by this name exist", and a run whose ``state.json`` is
|
|
2746
|
+
missing or corrupt still owns its control plane. Gating on state.json would
|
|
2747
|
+
sweep the counterpart out from under exactly the run an operator is trying to
|
|
2748
|
+
recover.
|
|
2749
|
+
|
|
2750
|
+
The two failures are distinguished because they mean opposite things. A
|
|
2751
|
+
*missing* runs dir is a real answer — no runs, so nothing is live — while an
|
|
2752
|
+
unreadable one answers nothing at all, and a sweep run against "no live names"
|
|
2753
|
+
would remove every state dir this project has. ``None`` is that second case.
|
|
2754
|
+
"""
|
|
2755
|
+
try:
|
|
2756
|
+
return {entry.name for entry in os.scandir(project / RUNS_DIR) if entry.is_dir()}
|
|
2757
|
+
except FileNotFoundError:
|
|
2758
|
+
return set()
|
|
2759
|
+
except OSError:
|
|
2760
|
+
return None
|
|
2761
|
+
|
|
2762
|
+
|
|
2763
|
+
def reconcile_orphan_state_dirs(project: Path, *, dry_run: bool = False) -> list[Path]:
|
|
2764
|
+
"""Remove this project's out-of-tree control-plane dirs whose run dir is gone.
|
|
2765
|
+
|
|
2766
|
+
The GC backstop for the events channel (#494). :func:`_discard_state_dir`
|
|
2767
|
+
removes the counterpart on every ordinary delete/archive, so this catches what
|
|
2768
|
+
that path could not: a run dir removed by hand or by an `rm -rf .froid-loop`,
|
|
2769
|
+
a delete that ran before this version existed, and any tail that failed
|
|
2770
|
+
quietly. Without it the state root accumulates one dead subtree per run
|
|
2771
|
+
forever, on a path outside the project that no operator thinks to look at.
|
|
2772
|
+
|
|
2773
|
+
Shaped like :func:`reconcile_orphan_worktrees`: enumerate on-disk truth,
|
|
2774
|
+
containment-test each path, remove with failures tolerated. Returns what was
|
|
2775
|
+
removed (or, under ``dry_run``, what would be).
|
|
2776
|
+
|
|
2777
|
+
Every path is built from an entry name this function itself enumerated —
|
|
2778
|
+
never from a caller-supplied ref, which is what :func:`_is_path_escape`
|
|
2779
|
+
refuses on the ref-resolution path. Entries that are not real directories are
|
|
2780
|
+
skipped, symlinks included: a link is not a state dir we created, and
|
|
2781
|
+
reporting one swept would be a false count even where ``rmtree`` refuses it.
|
|
2782
|
+
The containment test then covers what ``is_symlink`` cannot — a Windows
|
|
2783
|
+
*junction* reads as a plain directory while ``resolve()`` follows it, so
|
|
2784
|
+
without the test ``rmtree`` would empty a target sitting outside the root.
|
|
2785
|
+
That case is POSIX-invisible, and the tests say so rather than claim it.
|
|
2786
|
+
|
|
2787
|
+
Degrades to no-op rather than raising, in either direction: an underivable
|
|
2788
|
+
state root, an unreadable root, or an unreadable runs dir all sweep nothing.
|
|
2789
|
+
This is reclamation, not repair — leaving disk behind is the cheap outcome,
|
|
2790
|
+
and removing a live run's control plane is not.
|
|
2791
|
+
|
|
2792
|
+
Both guards hold ``RuntimeError`` alongside ``OSError`` for the same reason
|
|
2793
|
+
:func:`_discard_state_dir` does: every path here is resolved (the project by
|
|
2794
|
+
:func:`project_tag`, then the root, then each entry), and below 3.13
|
|
2795
|
+
``Path.resolve`` reports a symlink loop as ``RuntimeError``. A loop planted
|
|
2796
|
+
among the entries would otherwise escape a sweep whose whole contract is to
|
|
2797
|
+
degrade, and take the operator's ``clean`` down with it after its real work
|
|
2798
|
+
was already done.
|
|
2799
|
+
|
|
2800
|
+
**The two reads are ordered, and the order is the whole race guard.** State
|
|
2801
|
+
entries are enumerated *before* the live run-dir names, because a run creates
|
|
2802
|
+
its run dir strictly before its state dir — ``compose_run`` builds the
|
|
2803
|
+
``Journal`` (which mkdirs the run dir) and only then stamps the config digest
|
|
2804
|
+
(:func:`write_trusted_config_digest`, the earliest writer into the state dir
|
|
2805
|
+
since #498) and calls ``make_adapters``, whose ``SignalWatcher`` mkdirs the
|
|
2806
|
+
events dir alongside it. Reading entries first makes
|
|
2807
|
+
that ordering carry the guarantee: anything in ``entries`` had its state dir
|
|
2808
|
+
on disk at the first read, so its run dir was on disk *before* that, so the
|
|
2809
|
+
later ``live`` read is certain to contain it. Read the other way round, a run
|
|
2810
|
+
starting in the gap is missing from ``live`` and present in ``entries``, and
|
|
2811
|
+
an operator's ``clean`` deletes the control plane of a run that is starting
|
|
2812
|
+
right now — whose watcher then polls a primary that no longer exists, or
|
|
2813
|
+
simply never sees the Stop. A run dir that disappears *between* the reads is
|
|
2814
|
+
the opposite case and correctly swept: it is a real orphan by then.
|
|
2815
|
+
"""
|
|
2816
|
+
try:
|
|
2817
|
+
root = project_state_root(project)
|
|
2818
|
+
entries = sorted(root.iterdir())
|
|
2819
|
+
root_res = root.resolve()
|
|
2820
|
+
except (StateRootError, OSError, RuntimeError):
|
|
2821
|
+
return []
|
|
2822
|
+
live = _run_dir_names(project)
|
|
2823
|
+
if live is None:
|
|
2824
|
+
return []
|
|
2825
|
+
handled: list[Path] = []
|
|
2826
|
+
for entry in entries:
|
|
2827
|
+
if entry.name in live or entry.is_symlink() or not entry.is_dir():
|
|
2828
|
+
continue
|
|
2829
|
+
if entry.name == MUX_REGISTRY_DIR:
|
|
2830
|
+
# Not a run entry at all (`mux_registry_root`), and the one entry here
|
|
2831
|
+
# whose deletion costs more than the disk it reclaims: it holds the
|
|
2832
|
+
# `.port`/`.key` files every psmux verb resolves a session through, so
|
|
2833
|
+
# sweeping it while a server is up leaves that server alive,
|
|
2834
|
+
# unreachable, and invisible to `psmux ls` in any registry — the
|
|
2835
|
+
# manufactured orphan the root was moved out of the project tree to
|
|
2836
|
+
# avoid. Never reaped rather than reaped-when-empty: proving it empty
|
|
2837
|
+
# means asking every server in it whether it is alive, and this sweep
|
|
2838
|
+
# has no seam to the multiplexer (nor may it acquire one — it must
|
|
2839
|
+
# degrade to a no-op, and a transport probe cannot promise that).
|
|
2840
|
+
# psmux removes its own quartet on session shutdown, so what is left
|
|
2841
|
+
# behind is a directory of small files, not growth.
|
|
2842
|
+
continue
|
|
2843
|
+
try:
|
|
2844
|
+
entry.resolve().relative_to(root_res)
|
|
2845
|
+
except (OSError, RuntimeError, ValueError):
|
|
2846
|
+
continue
|
|
2847
|
+
handled.append(entry)
|
|
2848
|
+
if not dry_run:
|
|
2849
|
+
shutil.rmtree(entry, ignore_errors=True)
|
|
2850
|
+
return handled
|
|
2851
|
+
|
|
2852
|
+
|
|
2853
|
+
def _unlink_redirect(p: Path) -> None:
|
|
2854
|
+
"""Remove a link-like entry itself, never what it points at.
|
|
2855
|
+
|
|
2856
|
+
``shutil.rmtree`` REFUSES a directory symlink by design (it would otherwise
|
|
2857
|
+
delete the target's contents), and under ``ignore_errors=True`` that refusal
|
|
2858
|
+
is swallowed — so trimming a planted redirect reported success while leaving
|
|
2859
|
+
the link on disk. Unlink covers a POSIX symlink and a win32 file symlink;
|
|
2860
|
+
``rmdir`` is the win32 arm, where ``DeleteFileW`` rejects a directory symlink
|
|
2861
|
+
or junction and ``RemoveDirectoryW`` drops the reparse point without
|
|
2862
|
+
following it. Best-effort to match the ``rmtree`` beside it: a trim is
|
|
2863
|
+
reclamation, and a run dir we cannot fully reclaim is not a reason to abort
|
|
2864
|
+
the whole `clean`."""
|
|
2865
|
+
try:
|
|
2866
|
+
p.unlink()
|
|
2867
|
+
except OSError:
|
|
2868
|
+
with contextlib.suppress(OSError):
|
|
2869
|
+
p.rmdir()
|
|
2870
|
+
|
|
2871
|
+
|
|
2872
|
+
def trim_run_dir(run_dir: Path, *, dry_run: bool = False) -> list[Path]:
|
|
2873
|
+
"""Delete heavy scaffolding (the ``worktrees/`` tree and the retained
|
|
2874
|
+
verifier stream store) from a concluded run dir, preserving its TUI-visible
|
|
2875
|
+
core so the run still appears in the dashboard with full status/journal/logs.
|
|
2876
|
+
Returns the paths removed.
|
|
2877
|
+
|
|
2878
|
+
The run's out-of-tree control plane is deliberately left alone (see
|
|
2879
|
+
:func:`_discard_state_dir`): a trimmed run still exists and is still
|
|
2880
|
+
resumable, so its state dir has to outlive its scaffolding."""
|
|
2881
|
+
removed: list[Path] = []
|
|
2882
|
+
for p in heavy_run_entries(run_dir):
|
|
2883
|
+
link = is_link_like(p)
|
|
2884
|
+
if not (p.exists() or link):
|
|
2885
|
+
continue
|
|
2886
|
+
removed.append(p)
|
|
2887
|
+
if dry_run:
|
|
2888
|
+
continue
|
|
2889
|
+
if link:
|
|
2890
|
+
_unlink_redirect(p)
|
|
2891
|
+
else:
|
|
2892
|
+
shutil.rmtree(p, ignore_errors=True)
|
|
2893
|
+
return removed
|
|
2894
|
+
|
|
2895
|
+
|
|
2896
|
+
def _run_started_epoch(run_dir: Path) -> float | None:
|
|
2897
|
+
"""Unix time parsed from the run id's ``YYYYMMDD-HHMMSS`` prefix, or None
|
|
2898
|
+
when the name does not carry one (legacy/foreign id)."""
|
|
2899
|
+
try:
|
|
2900
|
+
return time.mktime(time.strptime(run_dir.name[:15], "%Y%m%d-%H%M%S"))
|
|
2901
|
+
except (ValueError, OverflowError):
|
|
2902
|
+
return None
|
|
2903
|
+
|
|
2904
|
+
|
|
2905
|
+
def runs_past_retention(
|
|
2906
|
+
run_dirs: list[Path], *, keep_n: int, keep_days: int = 0, now: float | None = None
|
|
2907
|
+
) -> list[Path]:
|
|
2908
|
+
"""The subset of ``run_dirs`` (oldest-first) beyond the retention window:
|
|
2909
|
+
not among the newest ``keep_n``, and — when ``keep_days`` is set — also older
|
|
2910
|
+
than ``keep_days`` days. ``keep_n <= 0`` retains nothing by count; an
|
|
2911
|
+
unparseable run id is treated as old enough to prune once past ``keep_n``."""
|
|
2912
|
+
ordered = list(run_dirs)
|
|
2913
|
+
candidates = (
|
|
2914
|
+
ordered[:-keep_n]
|
|
2915
|
+
if keep_n > 0 and len(ordered) > keep_n
|
|
2916
|
+
else ([] if keep_n > 0 else list(ordered))
|
|
2917
|
+
)
|
|
2918
|
+
if keep_days and keep_days > 0:
|
|
2919
|
+
cutoff = (time.time() if now is None else now) - keep_days * 86400
|
|
2920
|
+
return [rd for rd in candidates if (_run_started_epoch(rd) or 0.0) < cutoff]
|
|
2921
|
+
return candidates
|
|
2922
|
+
|
|
2923
|
+
|
|
2924
|
+
# ----------------------------------------------------------- escalation resolution
|
|
2925
|
+
|
|
2926
|
+
|
|
2927
|
+
class RearmError(Exception):
|
|
2928
|
+
"""The run/story is not in a re-armable escalation state."""
|
|
2929
|
+
|
|
2930
|
+
|
|
2931
|
+
def validate_restore_latch(
|
|
2932
|
+
state: RunState, task: StoryTask, story_key: str, *, worktree_isolation: bool = False
|
|
2933
|
+
) -> str | None:
|
|
2934
|
+
"""Every precondition an intent-gap patch-restore latch (Froid Plane #2564) must
|
|
2935
|
+
satisfy, in one place. Returns an operator-facing error string, or None to latch.
|
|
2936
|
+
|
|
2937
|
+
The single seam for both entry points: `rearm_escalation` (which performs the
|
|
2938
|
+
latch, and is also reachable programmatically — a TUI restore, a future caller)
|
|
2939
|
+
and `cli._resolve_restore_patch` (which fails fast *before* the interactive
|
|
2940
|
+
resolve session, so an unhonorable restore doesn't cost an agent conversation).
|
|
2941
|
+
Splitting these let a non-CLI caller bypass the worktree half; keeping them here
|
|
2942
|
+
means a caller cannot latch a patch the engine could never honor.
|
|
2943
|
+
|
|
2944
|
+
The CLI knows one thing this cannot: the *live* policy's isolation mode, which
|
|
2945
|
+
may have been edited between escalation and resolve. It passes that as
|
|
2946
|
+
`worktree_isolation`; the recorded `task.worktree_path` (how the unit actually
|
|
2947
|
+
executed) is checked here either way, so both entry points reject a
|
|
2948
|
+
worktree-isolation restore and the CLI additionally catches a policy flip.
|
|
2949
|
+
|
|
2950
|
+
Path resolution and trusted-roots containment stay CLI-side: they need
|
|
2951
|
+
`--project` and the loaded froid config, neither of which run state carries.
|
|
2952
|
+
"""
|
|
2953
|
+
# A sentinel-wedged story escalated BEFORE planning — there is no attempted
|
|
2954
|
+
# implementation to restore, and its re-arm re-dispatches a planning leg.
|
|
2955
|
+
# Keyed on the recorded detection verdict (task.sentinel_kind), not the on-disk
|
|
2956
|
+
# basename, mirroring rearm_escalation's sentinel-clear branch.
|
|
2957
|
+
if state.source == "stories" and task.sentinel_kind:
|
|
2958
|
+
return (
|
|
2959
|
+
f"story {story_key} is wedged on a pre-planning {task.sentinel_kind} sentinel — "
|
|
2960
|
+
"there is no attempted implementation to restore, and the re-drive starts "
|
|
2961
|
+
"at planning. Re-run resolve without a restore patch for a clean re-plan."
|
|
2962
|
+
)
|
|
2963
|
+
# Same seam, broader shape: a restore only works through the spec's in-review
|
|
2964
|
+
# flip, so an escalation with NO recorded spec (an ambiguous two-file wedge, an
|
|
2965
|
+
# unknown --story selector, a session that died before naming one) has no
|
|
2966
|
+
# routing target — the latch would stick, the flip would be skipped, and the
|
|
2967
|
+
# engine would lay the patch onto the tree before a planning leg.
|
|
2968
|
+
if not task.spec_file:
|
|
2969
|
+
return (
|
|
2970
|
+
f"story {story_key} has no recorded spec file, so a restored patch has no "
|
|
2971
|
+
"review to resume (the re-drive starts at planning). Re-run resolve "
|
|
2972
|
+
"without a restore patch for a from-scratch re-drive."
|
|
2973
|
+
)
|
|
2974
|
+
# Restore is an in-place-only recovery: a worktree-isolation re-drive discards
|
|
2975
|
+
# the unit's worktree (engine._finish_inflight — taking a patch saved inside it
|
|
2976
|
+
# along) and re-mounts a fresh one, so the re-apply could only fail on a
|
|
2977
|
+
# destroyed patch file. Reject up front instead of latching a patch that can
|
|
2978
|
+
# never restore.
|
|
2979
|
+
if worktree_isolation or task.worktree_path:
|
|
2980
|
+
return (
|
|
2981
|
+
"restore patch is unsupported for worktree-isolation runs (the re-drive "
|
|
2982
|
+
"discards and re-mounts the unit's worktree, so an in-place restore has "
|
|
2983
|
+
"nothing durable to land on) — re-arm from scratch instead: drop "
|
|
2984
|
+
"--restore-patch, or if the resolve agent recorded the restore in "
|
|
2985
|
+
"resolution.json, re-run with --no-interactive (which ignores that "
|
|
2986
|
+
"marker) instead of repeating the agent session"
|
|
2987
|
+
)
|
|
2988
|
+
return None
|
|
2989
|
+
|
|
2990
|
+
|
|
2991
|
+
def task_spec_path(task: StoryTask, state: RunState) -> Path:
|
|
2992
|
+
"""The recorded spec path, re-anchored on the tree it was persisted relative to.
|
|
2993
|
+
|
|
2994
|
+
`StoryTask._serialized_worktree_path` (`model.py`) persists a worktree-local spec
|
|
2995
|
+
RELATIVE to the mounted worktree root, and `from_dict` reads it back raw. Resolving
|
|
2996
|
+
that against the process cwd is not merely unreachable — it is actively wrong:
|
|
2997
|
+
`froid-loop resolve` runs from the project root, where the MAIN CHECKOUT carries the
|
|
2998
|
+
same `_froid-output/specs/...` layout, so a bare `Path(task.spec_file)` names the main
|
|
2999
|
+
checkout's copy of the story spec. `is_file()` then answers True, `confine_root`
|
|
3000
|
+
accepts it (it genuinely is under `project`), and the status flip and the baseline
|
|
3001
|
+
re-stamp both land on a file the run never used while the worktree's real spec is
|
|
3002
|
+
left on the escalated attempt's sha.
|
|
3003
|
+
|
|
3004
|
+
Absolute paths pass through: a spec outside the worktree is persisted verbatim.
|
|
3005
|
+
|
|
3006
|
+
Raises `ValueError` on an empty `task.spec_file` rather than documenting a
|
|
3007
|
+
precondition nothing enforces: `Path("")` is `.`, so `root / raw` would answer the
|
|
3008
|
+
ROOT DIRECTORY — a write target, not a spec. Every caller already guards; this is
|
|
3009
|
+
public now, so the next one gets an exception instead of a silent tree root.
|
|
3010
|
+
"""
|
|
3011
|
+
if not task.spec_file:
|
|
3012
|
+
raise ValueError("task_spec_path requires a non-empty task.spec_file")
|
|
3013
|
+
raw = Path(task.spec_file)
|
|
3014
|
+
if raw.is_absolute():
|
|
3015
|
+
return raw
|
|
3016
|
+
return task_spec_root(task, state) / raw
|
|
3017
|
+
|
|
3018
|
+
|
|
3019
|
+
def task_spec_root(task: StoryTask, state: RunState) -> Path:
|
|
3020
|
+
"""The tree a `task.spec_file` is anchored on — and confined to.
|
|
3021
|
+
|
|
3022
|
+
One definition backs both halves because they must not disagree: the root
|
|
3023
|
+
`task_spec_path` resolves against and the `confine_root` the writers validate the
|
|
3024
|
+
result against are the same claim about which tree owns this spec. Passing
|
|
3025
|
+
`state.project` while resolving against the worktree does not REFUSE the mismatch —
|
|
3026
|
+
`set_frontmatter_status`, `devcontract.strip_auto_run_result` and
|
|
3027
|
+
`verify.set_frontmatter_field` all answer an out-of-root path by silently dropping
|
|
3028
|
+
to the plain no-follow write, losing the confined arm's O_NOFOLLOW walk of the
|
|
3029
|
+
parent components (#593) with no signal at all.
|
|
3030
|
+
|
|
3031
|
+
Worktrees normally resolve under `<project>/.froid-loop/runs/...`, so the confined
|
|
3032
|
+
arm is taken by construction rather than by luck — no policy or env var can
|
|
3033
|
+
relocate them. The one escape is that `workspace.open_unit_workspace` stores a
|
|
3034
|
+
`.resolve()`d path: a symlinked `.froid-loop`, `runs` or `worktrees` lands the spec
|
|
3035
|
+
outside `project`, and before this anchor moved that silently degraded all three
|
|
3036
|
+
writes.
|
|
3037
|
+
|
|
3038
|
+
A worktree that CANNOT confine the anchored path yields the project instead. An
|
|
3039
|
+
absolute `spec_file` beside a set `worktree_path` is precisely the out-of-mount
|
|
3040
|
+
shape: `model._serialized_worktree_path` keeps a path verbatim exactly when
|
|
3041
|
+
`relative_to(worktree_path)` raises, so the two spellings did not share a prefix.
|
|
3042
|
+
Returning the worktree there would name a root that can never contain the path
|
|
3043
|
+
`task_spec_path` passes through — the three `_atomic_write_spec` writers would
|
|
3044
|
+
silently take the plain no-follow arm (losing #593's O_NOFOLLOW walk) and
|
|
3045
|
+
`_restore_rearmed_spec`, which calls the confined writer directly, would RAISE.
|
|
3046
|
+
|
|
3047
|
+
The project can often confine it. Where nothing can, the THREE `_atomic_write_spec`
|
|
3048
|
+
writers land on the arm they already took — they select lexically, so an out-of-root
|
|
3049
|
+
path simply takes the plain no-follow write as before. That is not true of every
|
|
3050
|
+
writer: `_restore_rearmed_spec` calls `atomic_write_bytes_confined` DIRECTLY with no
|
|
3051
|
+
lexical arm, so for a spec outside both the mount and the project — the shared
|
|
3052
|
+
artifact dir `_spec_is_shared_with_the_redrive` treats as first-class and reachable —
|
|
3053
|
+
it raises `UnconfinedWriteError` and the re-arm's undo is lost with the spec already
|
|
3054
|
+
flipped and stripped. That asymmetry PRE-DATES this anchor (the previous body
|
|
3055
|
+
returned the worktree there, which equally cannot confine the path) and is tracked
|
|
3056
|
+
separately; it is named here so the paragraph is not read as covering it.
|
|
3057
|
+
|
|
3058
|
+
The arm is not unconditionally an improvement either, and that exception is graded by
|
|
3059
|
+
`test_task_spec_root_refuses_a_spec_the_project_cannot_reach`: `_atomic_write_spec`
|
|
3060
|
+
picks its arm on a LEXICAL `is_relative_to`, but the confined arm it picks then
|
|
3061
|
+
walks the components below the root and refuses a redirect (`open_dir_confined` on
|
|
3062
|
+
POSIX, `path_is_confined` on win32). A spec that is lexically under the project but
|
|
3063
|
+
reached THROUGH a symlinked component — a symlinked `_froid-output`, say — therefore
|
|
3064
|
+
moves from a succeeding plain no-follow write to `UnconfinedWriteError`, which
|
|
3065
|
+
`rearm_escalation` re-raises as `RearmError`. That is a re-arm which used to
|
|
3066
|
+
complete and now aborts, so this arm is not the pure improvement an earlier draft of
|
|
3067
|
+
this docstring claimed.
|
|
3068
|
+
|
|
3069
|
+
It is kept anyway, because the alternative is worse. Predicting the walk here (gate
|
|
3070
|
+
the arm on `path_is_confined` and fall back to the worktree) makes the ROOT depend
|
|
3071
|
+
on filesystem state: `path_is_confined` answers False for a component it cannot
|
|
3072
|
+
probe, so a spec whose parent does not exist yet would anchor on the worktree and
|
|
3073
|
+
the same spec would anchor on the project once the directory appeared. A confine
|
|
3074
|
+
root that moves under a `mkdir` is not a definition. The refusal is also the correct
|
|
3075
|
+
posture on its own terms — #593 exists to refuse writes through a link on a path
|
|
3076
|
+
that came from a session-driven scan — so this trades a narrow, LOUD failure for a
|
|
3077
|
+
deterministic rule, and the failure names the path in its message.
|
|
3078
|
+
|
|
3079
|
+
The test is the same lexical `is_relative_to` the writer gates on, so the root and
|
|
3080
|
+
the writer's ARM SELECTION agree by construction; only the walk below can still
|
|
3081
|
+
refuse. Deliberately not canonicalized: `_spec_is_shared_with_the_redrive` answers a
|
|
3082
|
+
DIFFERENT question (is this spec reachable by the re-drive) and canonicalizes for
|
|
3083
|
+
it, but matching that here would diverge from the gate this value is measured
|
|
3084
|
+
against and change writes that are correct today.
|
|
3085
|
+
"""
|
|
3086
|
+
worktree = task.worktree_path
|
|
3087
|
+
if not worktree:
|
|
3088
|
+
return Path(state.project)
|
|
3089
|
+
raw = Path(task.spec_file or "")
|
|
3090
|
+
if raw.is_absolute() and not raw.is_relative_to(worktree):
|
|
3091
|
+
return Path(state.project)
|
|
3092
|
+
return Path(worktree)
|
|
3093
|
+
|
|
3094
|
+
|
|
3095
|
+
def task_stories_root(task: StoryTask | None, state: RunState) -> Path:
|
|
3096
|
+
"""The tree this run's STORIES FOLDER lives in — the workspace root, not a
|
|
3097
|
+
confinement root.
|
|
3098
|
+
|
|
3099
|
+
Deliberately NOT `task_spec_root`, which the sentinel and stories-block readers
|
|
3100
|
+
used to borrow. That function answers "which tree can CONFINE a write to
|
|
3101
|
+
`task.spec_file`", and its out-of-mount arm falls back to the project precisely so
|
|
3102
|
+
a `confine_root` can never fail to contain the anchored path. Reusing that answer
|
|
3103
|
+
here imported a write-confinement decision into a READ of a different file: for an
|
|
3104
|
+
isolated run whose `spec_file` is absolute and lexically outside the mount — the
|
|
3105
|
+
shape `model._serialized_worktree_path` persists verbatim, reachable whenever a
|
|
3106
|
+
symlinked component makes a spec that physically lives in the mount look outside
|
|
3107
|
+
it, since `verify.resolve_spec_path` deliberately does not `.resolve()` — the
|
|
3108
|
+
stories folder would be looked up in the MAIN CHECKOUT while
|
|
3109
|
+
`stories_engine._stories_folder` answers the worktree for the same task. One
|
|
3110
|
+
surface would then describe two trees, which is the exact defect the spec anchor
|
|
3111
|
+
exists to close.
|
|
3112
|
+
|
|
3113
|
+
So this mirrors `_stories_folder`'s own rule instead: the mount whenever the task
|
|
3114
|
+
holds one, the project otherwise. `spec_file` does not enter into it — the stories
|
|
3115
|
+
folder is located by `state.spec_folder` relative to the workspace root, and a
|
|
3116
|
+
task's spec being elsewhere says nothing about where its story manifest lives.
|
|
3117
|
+
|
|
3118
|
+
A mount that is GONE degrades to the project. `worktree_path` is cleared at
|
|
3119
|
+
exactly one site in the engine — the restart discard — so a task that reached a
|
|
3120
|
+
terminal phase through successful integration keeps naming the unit worktree its
|
|
3121
|
+
teardown already removed. The `done_checkpoint` pause is raised in precisely that
|
|
3122
|
+
window, and the TUI reads this for the checkpoint card's title and description, so
|
|
3123
|
+
trusting the stale field lost the committed story's manifest to a deleted
|
|
3124
|
+
directory while the merged copy sat in the project checkout.
|
|
3125
|
+
|
|
3126
|
+
Answering on filesystem state is right HERE and would be wrong in
|
|
3127
|
+
`task_spec_root`: that one is a write-confinement root, where a value that moves
|
|
3128
|
+
under a `mkdir` is not a definition. This is a READ locator, and observation
|
|
3129
|
+
degrades rather than raising — a probe that cannot answer falls back to the tree
|
|
3130
|
+
that always exists.
|
|
3131
|
+
|
|
3132
|
+
Accepts `None` so the two call sites do not each re-spell the no-task fallback.
|
|
3133
|
+
"""
|
|
3134
|
+
if task is None or not task.worktree_path:
|
|
3135
|
+
return Path(state.project)
|
|
3136
|
+
mount = Path(task.worktree_path)
|
|
3137
|
+
try:
|
|
3138
|
+
if not mount.is_dir():
|
|
3139
|
+
return Path(state.project)
|
|
3140
|
+
except OSError:
|
|
3141
|
+
return Path(state.project)
|
|
3142
|
+
return mount
|
|
3143
|
+
|
|
3144
|
+
|
|
3145
|
+
def _spec_is_shared_with_the_redrive(state: RunState, task: StoryTask) -> bool:
|
|
3146
|
+
"""True when the recorded spec lives outside BOTH checkouts, so the re-arm's status
|
|
3147
|
+
flip survives a mount's disposal and the ISOLATED re-drive reads it.
|
|
3148
|
+
|
|
3149
|
+
Asked only of a re-drive that will mount (`spec_reaches_the_redrive`'s isolated
|
|
3150
|
+
arm), and deliberately not of a task that HAS a mount: those are two different
|
|
3151
|
+
questions, and a policy flip separates them. A run switched from `isolation = "none"`
|
|
3152
|
+
to `"worktree"` while an escalation is paused re-drives isolated with no mount
|
|
3153
|
+
recorded at all, and the recorded spec is then measured against the project alone —
|
|
3154
|
+
which is the whole point, since the fresh worktree is cut from git and reads no
|
|
3155
|
+
working tree.
|
|
3156
|
+
|
|
3157
|
+
The case: artifact dirs configured outside the project tree. `ProjectPaths.rebased`
|
|
3158
|
+
leaves those exactly where they are ("configured outside the project tree; doesn't
|
|
3159
|
+
move") — they are SHARED across checkouts, not per-worktree — so the spec the dev
|
|
3160
|
+
session reported resolves to one file that every worktree sees. The re-drive reads it
|
|
3161
|
+
back through `verify.resolve_spec_path`, whose absolute branch passes the value
|
|
3162
|
+
through untouched, and `engine._dispatched_spec_for_attempt` then accepts it because
|
|
3163
|
+
the rebased `implementation_artifacts` is still that same external directory.
|
|
3164
|
+
|
|
3165
|
+
Both roots are load-bearing, and neither implies the other:
|
|
3166
|
+
|
|
3167
|
+
- INSIDE the worktree — the file the fresh mount destroys. Unreachable.
|
|
3168
|
+
- inside the PROJECT but outside the worktree — the main checkout's copy. The write
|
|
3169
|
+
lands, but the re-drive cannot use it: under isolation `workspace.paths` is rebased
|
|
3170
|
+
onto the fresh worktree, so `verify.spec_within_roots` measures the main
|
|
3171
|
+
checkout's path against worktree-local roots and rejects it. Unreachable, and this
|
|
3172
|
+
is the one shape the worktree test alone would wrongly exempt.
|
|
3173
|
+
- outside both — the shared artifact dir above. Reachable.
|
|
3174
|
+
|
|
3175
|
+
(The two are not nested: worktrees normally sit under `<project>/.froid-loop/runs/`,
|
|
3176
|
+
but `workspace.open_unit_workspace` stores a `.resolve()`d path, so a symlinked
|
|
3177
|
+
`.froid-loop` puts the mount outside the project.)
|
|
3178
|
+
|
|
3179
|
+
The recorded spelling opens the question but does not answer it.
|
|
3180
|
+
`StoryTask._serialized_worktree_path` persists a spec RELATIVE whenever it sits under
|
|
3181
|
+
the mounted worktree (and, with no mount, whenever the run recorded it relative to
|
|
3182
|
+
the project), and verbatim (absolute) otherwise — so an absolute value is the only
|
|
3183
|
+
shape that can be shared. But that relativize is a LEXICAL `relative_to` against the
|
|
3184
|
+
same `worktree_path` read here, so all an absolute value proves is that the two
|
|
3185
|
+
spellings did not share a prefix. A spec reported through a symlink or a `..` segment
|
|
3186
|
+
sits inside the worktree and is persisted absolute all the same, and answering
|
|
3187
|
+
"shared" for it would suppress the warning on a spec that really is destroyed with
|
|
3188
|
+
the worktree.
|
|
3189
|
+
|
|
3190
|
+
So containment is decided on the CANONICAL paths, and a host that cannot canonicalize
|
|
3191
|
+
one of them answers "not shared". That degrade is the safe direction and the reason
|
|
3192
|
+
this does not use `resolve_or_lexical`: its fallback is `absolute()`, which does not
|
|
3193
|
+
fold `..`, so a spec spelled through either checkout would come back looking external
|
|
3194
|
+
and go silent — trading a wrong warning for no warning at all."""
|
|
3195
|
+
raw = Path(task.spec_file or "")
|
|
3196
|
+
if not raw.is_absolute():
|
|
3197
|
+
return False
|
|
3198
|
+
try:
|
|
3199
|
+
# the house pair — `resolve()` raises RuntimeError, not OSError, for a symlink
|
|
3200
|
+
# loop on the 3.11/3.12 floor
|
|
3201
|
+
real = raw.resolve()
|
|
3202
|
+
if real.is_relative_to(Path(state.project).resolve()):
|
|
3203
|
+
return False
|
|
3204
|
+
if task.worktree_path and real.is_relative_to(Path(task.worktree_path).resolve()):
|
|
3205
|
+
return False
|
|
3206
|
+
return True
|
|
3207
|
+
except (OSError, RuntimeError):
|
|
3208
|
+
return False
|
|
3209
|
+
|
|
3210
|
+
|
|
3211
|
+
def _spec_is_inside_the_mount(task: StoryTask) -> bool:
|
|
3212
|
+
"""True when the file `task_spec_path` names sits INSIDE the mount this task
|
|
3213
|
+
recorded — so a write to it cannot reach an IN-PLACE re-drive, which reads the main
|
|
3214
|
+
checkout.
|
|
3215
|
+
|
|
3216
|
+
The mirror of `_spec_is_shared_with_the_redrive`, for the other arm of
|
|
3217
|
+
`spec_reaches_the_redrive`. Reachable only through a policy flip: a run switched
|
|
3218
|
+
from `isolation = "worktree"` to `"none"` while an escalation is paused still
|
|
3219
|
+
carries the escalated attempt's `worktree_path`, so `task_spec_path` re-anchors the
|
|
3220
|
+
edit on that mount while `engine._run_story` re-runs the story in the main checkout.
|
|
3221
|
+
`_finish_inflight` releases the mount-owned spelling at RESUME, which is after
|
|
3222
|
+
`froid-loop resolve` has already written the context and re-armed — this is what the
|
|
3223
|
+
human and the agent are told in the meantime.
|
|
3224
|
+
|
|
3225
|
+
Unlike the shared test, containment inside the PROJECT is not disqualifying: an
|
|
3226
|
+
in-place re-drive reads the main checkout's working tree, so a spec anywhere the
|
|
3227
|
+
project can see it reaches. Only the mount is out of reach.
|
|
3228
|
+
|
|
3229
|
+
A relative spelling beside a recorded mount is inside it BY CONSTRUCTION —
|
|
3230
|
+
`_serialized_worktree_path` relativizes exactly when `relative_to(worktree_path)`
|
|
3231
|
+
succeeds — so it needs no filesystem probe and gets none. Absolute spellings are
|
|
3232
|
+
canonicalized for the same reason the shared test does it (a `..` segment or a
|
|
3233
|
+
symlinked component puts a physically-inside path outside lexically), and a host
|
|
3234
|
+
that cannot canonicalize degrades to "inside": the safe direction here is the one
|
|
3235
|
+
that WARNS, matching the shared test's own degrade.
|
|
3236
|
+
"""
|
|
3237
|
+
if not task.worktree_path:
|
|
3238
|
+
return False
|
|
3239
|
+
raw = Path(task.spec_file or "")
|
|
3240
|
+
if not raw.is_absolute():
|
|
3241
|
+
return True
|
|
3242
|
+
try:
|
|
3243
|
+
return raw.resolve().is_relative_to(Path(task.worktree_path).resolve())
|
|
3244
|
+
except (OSError, RuntimeError):
|
|
3245
|
+
return True
|
|
3246
|
+
|
|
3247
|
+
|
|
3248
|
+
def redrive_base_ref(state: RunState, *, isolated_redrive: bool) -> str:
|
|
3249
|
+
"""The ref whose committed tree the re-drive will actually read this unit's spec
|
|
3250
|
+
from: the run's PINNED `target_branch` when the re-drive will MOUNT, ``HEAD``
|
|
3251
|
+
otherwise.
|
|
3252
|
+
|
|
3253
|
+
Not `HEAD` in both cases, because the isolated re-drive never reads the main
|
|
3254
|
+
checkout's working ref. `engine._finish_inflight` discards the escalated worktree
|
|
3255
|
+
and its branch and `_run_story` mounts a replacement, and
|
|
3256
|
+
`workspace.open_unit_workspace` cuts that fresh branch from the `base` it is handed
|
|
3257
|
+
— `worktree_flow.run_isolated` passes `state.target_branch`, pinned once at run
|
|
3258
|
+
start so resume keeps targeting the same branch. An operator who checks out another
|
|
3259
|
+
branch in the main checkout while the escalation is paused therefore moves `HEAD`
|
|
3260
|
+
off the tree the re-drive reads, in either direction: a correction committed on the
|
|
3261
|
+
now-current branch is invisible to the re-drive, and one committed on the target
|
|
3262
|
+
branch is invisible to `HEAD`.
|
|
3263
|
+
|
|
3264
|
+
That mattered once `rearm-spec-write-unreachable` began holding the resume
|
|
3265
|
+
(`rearm_holds_the_resume`): reading the wrong ref does not merely mis-word a
|
|
3266
|
+
warning, it either resumes a re-drive that re-wedges on the target branch's
|
|
3267
|
+
terminal status, or holds a resume whose work is already committed where the
|
|
3268
|
+
re-drive will find it.
|
|
3269
|
+
|
|
3270
|
+
`isolated_redrive` is the LIVE policy's isolation mode, injected by the caller, and
|
|
3271
|
+
the task drops out of the signature entirely. It used to be inferred from
|
|
3272
|
+
`task.worktree_path` — a recorded mount — and that is the retrospective fact, not
|
|
3273
|
+
this one. `engine._run_story` selects the mode from `self._isolated` alone, and an
|
|
3274
|
+
isolation change mid-run is journalled, never refused, so the recorded mount and the
|
|
3275
|
+
next re-drive part company in BOTH directions: a run flipped to `"none"` still
|
|
3276
|
+
carries the escalated attempt's mount and would name the pinned branch for an
|
|
3277
|
+
in-place re-drive that reads `HEAD`, and one flipped to `"worktree"` carries no
|
|
3278
|
+
mount at all and would name `HEAD` for a re-drive that mounts. Both send a
|
|
3279
|
+
correction to a tree the run does not read. The same injection is how
|
|
3280
|
+
`validate_restore_latch` already learns this fact.
|
|
3281
|
+
|
|
3282
|
+
That the caller must supply it is the point: `froid-loop resolve` computes this
|
|
3283
|
+
context in a SEPARATE process, before the resume ever runs, so no amount of
|
|
3284
|
+
resume-time bookkeeping on `task.worktree_path` could have reached it. The fact
|
|
3285
|
+
enters the pure core as a parameter and nothing here reads policy.
|
|
3286
|
+
|
|
3287
|
+
An empty `target_branch` beside an isolated re-drive is a MISSING value, not a
|
|
3288
|
+
divergent one: `ensure_target_branch` pins the field before any worktree mounts, so
|
|
3289
|
+
only a state.json predating it can reach here, and that shape degrades to exactly
|
|
3290
|
+
the ref it read before — the same migration `restamp_code_root` gives an unrecorded
|
|
3291
|
+
root. Answering ``""`` instead would hold the resume on a per-configuration
|
|
3292
|
+
constant, the failure the record's narrowing exists to avoid.
|
|
3293
|
+
"""
|
|
3294
|
+
if isolated_redrive and state.target_branch:
|
|
3295
|
+
return state.target_branch
|
|
3296
|
+
return "HEAD"
|
|
3297
|
+
|
|
3298
|
+
|
|
3299
|
+
def spec_reaches_the_redrive(task: StoryTask, state: RunState, *, isolated_redrive: bool) -> bool:
|
|
3300
|
+
"""Whether an edit to this task's spec survives to the re-drive that reads it.
|
|
3301
|
+
|
|
3302
|
+
The other half of `task_spec_path`'s answer, and the two ask different questions of
|
|
3303
|
+
different sources. That one is RETROSPECTIVE — which tree owns the state this task
|
|
3304
|
+
already persisted — and reads the recorded mount, correctly. This one is
|
|
3305
|
+
PROSPECTIVE, so it reads `isolated_redrive`: the live policy's mode, injected by the
|
|
3306
|
+
caller exactly as `redrive_base_ref` and `validate_restore_latch` take it.
|
|
3307
|
+
|
|
3308
|
+
Both arms are about the same gap between where the edit LANDS (`task_spec_path`) and
|
|
3309
|
+
where the re-drive READS:
|
|
3310
|
+
|
|
3311
|
+
- the re-drive will MOUNT: it reads the COMMITTED tree of a fresh worktree, so only
|
|
3312
|
+
a spec outside both checkouts is one file they share
|
|
3313
|
+
(`_spec_is_shared_with_the_redrive` carries that argument in full). True whether
|
|
3314
|
+
or not a mount is recorded — a run flipped to `isolation = "worktree"` mid-pause
|
|
3315
|
+
has none, and its working-tree edit vanishes just as silently.
|
|
3316
|
+
- the re-drive runs IN PLACE: it reads the main checkout's working tree, so the edit
|
|
3317
|
+
reaches unless it landed inside a recorded mount (`_spec_is_inside_the_mount`) —
|
|
3318
|
+
the flip in the other direction.
|
|
3319
|
+
|
|
3320
|
+
Public because `resolve.build_context` needs it for the same reason
|
|
3321
|
+
`rearm_escalation` does: the context hands a human and an agent a `spec_file` to
|
|
3322
|
+
edit, and an edit to a doomed copy is worse than no edit — it looks like it landed.
|
|
3323
|
+
"""
|
|
3324
|
+
if isolated_redrive:
|
|
3325
|
+
return _spec_is_shared_with_the_redrive(state, task)
|
|
3326
|
+
return not _spec_is_inside_the_mount(task)
|
|
3327
|
+
|
|
3328
|
+
|
|
3329
|
+
def _upstream_artifacts_folder(state: RunState) -> Path:
|
|
3330
|
+
"""The folder holding the UPSTREAM stories artifacts a sentinel's correction goes
|
|
3331
|
+
into — anchored on the project, never on a mount.
|
|
3332
|
+
|
|
3333
|
+
Deliberately NOT `task_stories_root`, which answers "which tree does this RUN read
|
|
3334
|
+
its manifest out of" and is the mount whenever the task holds one. This answers
|
|
3335
|
+
"which folder does the CORRECTION land in", and `resolve.run_session` settles that
|
|
3336
|
+
independently of the mount: the agent runs with `cwd=project` and the artifacts are
|
|
3337
|
+
named by a project-relative `state.spec_folder`, so the writes go to the main
|
|
3338
|
+
checkout even for a task that recorded a worktree. An absolute `spec_folder` — the
|
|
3339
|
+
external-artifact-dir layout `[stories] source` allows — is left where it is, which
|
|
3340
|
+
is what `resolve_spec_folder` already does and what makes it shared across
|
|
3341
|
+
checkouts.
|
|
3342
|
+
|
|
3343
|
+
One locator for all three consumers (the gate, the proof, and the journal record)
|
|
3344
|
+
so a record can never name a folder its own gate did not measure.
|
|
3345
|
+
"""
|
|
3346
|
+
from .stories import resolve_spec_folder
|
|
3347
|
+
|
|
3348
|
+
return resolve_spec_folder(Path(state.project), state.spec_folder)
|
|
3349
|
+
|
|
3350
|
+
|
|
3351
|
+
def stories_reach_the_redrive(task: StoryTask, state: RunState, *, isolated_redrive: bool) -> bool:
|
|
3352
|
+
"""Whether an edit to this run's UPSTREAM stories artifacts survives to the re-drive.
|
|
3353
|
+
|
|
3354
|
+
`spec_reaches_the_redrive` asked of `SPEC.md` / `stories.yaml` instead of the frozen
|
|
3355
|
+
spec, for the one wedge where the spec is not the artifact being corrected: a
|
|
3356
|
+
fixed-slug pre-planning-halt SENTINEL. A sentinel is cleared by DELETION, so the
|
|
3357
|
+
re-arm drops `task.spec_file` and there is no spec write whose reachability that
|
|
3358
|
+
helper could measure — which is why its arm is an `else` this path never entered,
|
|
3359
|
+
and why no hold ever fired for a sentinel. But the correction that stops the
|
|
3360
|
+
sentinel RECURRING is upstream, in the artifacts `froid-loop-resolve/SKILL.md` sends
|
|
3361
|
+
the agent to instead of the sentinel, and it faces the identical gap: an isolated
|
|
3362
|
+
re-drive mounts fresh from `redrive_base_ref` and re-plans from a COMMITTED tree, so
|
|
3363
|
+
an uncommitted upstream edit is invisible and the re-plan mints the sentinel again.
|
|
3364
|
+
|
|
3365
|
+
The two arms are NOT the spec question's, and the difference is where the write
|
|
3366
|
+
lands. `task_spec_path` re-anchors a spec write ON the recorded mount, so a policy
|
|
3367
|
+
flip separates writer from reader in BOTH directions. The upstream artifacts are
|
|
3368
|
+
named by a project-relative `state.spec_folder` and `resolve.run_session` runs the
|
|
3369
|
+
agent with `cwd=project`, so the correction lands in the MAIN CHECKOUT whichever way
|
|
3370
|
+
the flip went. That collapses one arm:
|
|
3371
|
+
|
|
3372
|
+
- the re-drive runs IN PLACE: it reads the main checkout's working tree —
|
|
3373
|
+
`stories_engine._stories_folder` anchors a relative folder on the live workspace
|
|
3374
|
+
root, which is the project under `isolation = "none"`. Writer and reader are the
|
|
3375
|
+
same tree, so the edit reaches. The recorded mount does not enter into it; a run
|
|
3376
|
+
flipped `"worktree" -> "none"` mid-pause still carries one, and it is not where
|
|
3377
|
+
the correction went.
|
|
3378
|
+
- the re-drive will MOUNT: the fresh worktree is cut from git and checks out TRACKED
|
|
3379
|
+
files, so no working-tree write reaches it — with the single exception
|
|
3380
|
+
`_spec_is_shared_with_the_redrive` carries in full, an artifact dir configured
|
|
3381
|
+
OUTSIDE the project tree, which `ProjectPaths.rebased` leaves exactly where it is
|
|
3382
|
+
and every worktree therefore reads through the same absolute path. True whether or
|
|
3383
|
+
not a mount is recorded: a run flipped `"none" -> "worktree"` has none, and its
|
|
3384
|
+
working-tree edit vanishes just as silently.
|
|
3385
|
+
|
|
3386
|
+
Both roots are tested on the mounting arm for the same reason that helper tests
|
|
3387
|
+
both: worktrees normally sit under `<project>/.froid-loop/runs/`, but
|
|
3388
|
+
`workspace.open_unit_workspace` stores a `.resolve()`d path, so a symlinked
|
|
3389
|
+
`.froid-loop` puts the mount outside the project and "outside the project" alone
|
|
3390
|
+
would not be "shared".
|
|
3391
|
+
|
|
3392
|
+
Canonicalized because a `..` segment or a symlinked component puts a
|
|
3393
|
+
physically-inside path outside lexically, and a host that cannot canonicalize
|
|
3394
|
+
degrades to UNREACHABLE — the direction that warns, matching the degrade both spec
|
|
3395
|
+
helpers already chose.
|
|
3396
|
+
"""
|
|
3397
|
+
if not isolated_redrive:
|
|
3398
|
+
return True
|
|
3399
|
+
try:
|
|
3400
|
+
real = _upstream_artifacts_folder(state).resolve()
|
|
3401
|
+
if real.is_relative_to(Path(state.project).resolve()):
|
|
3402
|
+
return False
|
|
3403
|
+
if task.worktree_path and real.is_relative_to(Path(task.worktree_path).resolve()):
|
|
3404
|
+
return False
|
|
3405
|
+
return True
|
|
3406
|
+
except (OSError, RuntimeError):
|
|
3407
|
+
return False
|
|
3408
|
+
|
|
3409
|
+
|
|
3410
|
+
# The two upstream artifacts `froid-loop-resolve/SKILL.md` names for a sentinel wedge:
|
|
3411
|
+
# the epic spec and the story manifest the planner reads. Fixed names, discovered as
|
|
3412
|
+
# siblings in the spec folder (`stories.STORIES_FILENAME`'s own docstring says so), so
|
|
3413
|
+
# the proof below can name them without parsing anything.
|
|
3414
|
+
_UPSTREAM_ARTIFACTS = ("SPEC.md", "stories.yaml")
|
|
3415
|
+
|
|
3416
|
+
|
|
3417
|
+
def _redrive_reads_the_upstream_artifacts(state: RunState) -> bool:
|
|
3418
|
+
"""PROOF that the tree the re-drive re-plans from already carries this checkout's
|
|
3419
|
+
upstream artifacts byte-for-byte. ``False`` on every uncertainty.
|
|
3420
|
+
|
|
3421
|
+
`_redrive_spec_status`'s counterpart for the sentinel path, and it exists for the
|
|
3422
|
+
same reason: without it the record its caller writes is a per-configuration
|
|
3423
|
+
CONSTANT. Every isolated stories run resolves its spec folder inside the project,
|
|
3424
|
+
so `stories_reach_the_redrive` answers "unreachable" for 100% of sentinel re-arms
|
|
3425
|
+
under `isolation = "worktree"` — and that record now HOLDS THE RESUME
|
|
3426
|
+
(`rearm_holds_the_resume`), so an unnarrowed gate would not merely train the
|
|
3427
|
+
operator to scroll past a warning, it would turn every one of those re-arms into a
|
|
3428
|
+
two-command gesture for an outcome nothing decided. That is the exact failure the
|
|
3429
|
+
spec arm's own narrowing exists to avoid, and it is worse here.
|
|
3430
|
+
|
|
3431
|
+
There is no status to read for a sentinel — it is cleared by deletion and the
|
|
3432
|
+
re-plan routes on nothing — so the proof is byte equality instead: if the ref the
|
|
3433
|
+
fresh worktree is cut from already holds what this checkout holds, the re-drive
|
|
3434
|
+
re-plans from exactly the tree the operator is looking at and there is nothing left
|
|
3435
|
+
to commit. If it does not, the operator has upstream work the re-drive will not read.
|
|
3436
|
+
|
|
3437
|
+
Read at `redrive_base_ref` and NOT at the code root's `HEAD`, for the reason that
|
|
3438
|
+
function documents: an operator who checks out another branch while the escalation
|
|
3439
|
+
is paused moves `HEAD` off the tree the re-drive reads, in either direction. It is
|
|
3440
|
+
asked for the MOUNTING mode unconditionally, and takes no `isolated_redrive` to say
|
|
3441
|
+
so, because there is exactly one reachable caller and it has already established
|
|
3442
|
+
that: `stories_reach_the_redrive` answers "reaches" for every in-place re-drive, so
|
|
3443
|
+
the `and` short-circuits before this runs. Carrying a second in-place arm here would
|
|
3444
|
+
not be defence in depth — it would SHADOW that one, leaving the reachability arm
|
|
3445
|
+
ungraded by any test and a wrong answer there invisible.
|
|
3446
|
+
|
|
3447
|
+
Every uncertainty answers ``False`` so the record fires and the resume holds: a
|
|
3448
|
+
folder outside the code root (which includes the external artifact dir, already
|
|
3449
|
+
exempted one gate earlier as SHARED), an unreadable working-tree file, an untracked
|
|
3450
|
+
or non-blob path at that ref (the read answers ``None``, which no byte string
|
|
3451
|
+
equals), or any `GitError` — including the project simply not being a repository. Suppression requires proof that the work is already done.
|
|
3452
|
+
|
|
3453
|
+
The blob is materialized through `worktree_file_bytes_at_revision`, not read raw:
|
|
3454
|
+
that function exists for precisely this comparison — a live checkout file against
|
|
3455
|
+
its committed counterpart — because Git's smudge, EOL and working-tree-encoding
|
|
3456
|
+
filters mean a byte-exact LF blob is legitimately a CRLF file on disk under
|
|
3457
|
+
`core.autocrlf=true`. Comparing raw blob bytes would mismatch every artifact on a
|
|
3458
|
+
Windows checkout and re-create, on one platform, the constant this narrowing exists
|
|
3459
|
+
to prevent.
|
|
3460
|
+
"""
|
|
3461
|
+
base = _upstream_artifacts_folder(state)
|
|
3462
|
+
code_root = state.code_root
|
|
3463
|
+
ref = redrive_base_ref(state, isolated_redrive=True)
|
|
3464
|
+
for name in _UPSTREAM_ARTIFACTS:
|
|
3465
|
+
live = base / name
|
|
3466
|
+
try:
|
|
3467
|
+
rel = live.relative_to(code_root).as_posix()
|
|
3468
|
+
except ValueError:
|
|
3469
|
+
return False
|
|
3470
|
+
try:
|
|
3471
|
+
committed = verify.worktree_file_bytes_at_revision(code_root, ref, rel)
|
|
3472
|
+
except verify.GitError:
|
|
3473
|
+
return False
|
|
3474
|
+
try:
|
|
3475
|
+
working = live.read_bytes()
|
|
3476
|
+
except OSError:
|
|
3477
|
+
return False
|
|
3478
|
+
if committed != working:
|
|
3479
|
+
return False
|
|
3480
|
+
return True
|
|
3481
|
+
|
|
3482
|
+
|
|
3483
|
+
def _restore_rearmed_spec(
|
|
3484
|
+
spec_path: Path, original: bytes | None, task: StoryTask, state: RunState
|
|
3485
|
+
) -> None:
|
|
3486
|
+
"""Put back the bytes a re-arm FOUND on the spec, for the aborts that can fire after
|
|
3487
|
+
a write has already landed.
|
|
3488
|
+
|
|
3489
|
+
`rearm_escalation` holds an invariant its own refusals depend on: an aborted re-arm
|
|
3490
|
+
leaves the spec byte-identical, so the escalation stays armed and the human can fix
|
|
3491
|
+
the file and re-run resolve. TWO of its four refusals earn that by SEQUENCING alone —
|
|
3492
|
+
the flip's read-back check and the `FrontmatterWriteError` arm both raise before
|
|
3493
|
+
`devcontract.strip_auto_run_result` runs, which is why that strip is deliberately
|
|
3494
|
+
ordered after them, and `set_frontmatter_status` decides it cannot move a `status:`
|
|
3495
|
+
before it writes anything. The other two cannot be sequenced out of the hazard, and
|
|
3496
|
+
both call this:
|
|
3497
|
+
|
|
3498
|
+
* The baseline re-stamp needs `task.baseline_commit` from the advance, and the
|
|
3499
|
+
advance must itself run after the spec block (a just-cleared stories sentinel would
|
|
3500
|
+
otherwise be captured into `baseline_untracked` as phantom pre-existing residue).
|
|
3501
|
+
* The `(OSError, UnicodeDecodeError)` arm spans BOTH spec helpers, and the strip is
|
|
3502
|
+
the later one — a fault raised inside it is raised after the flip published.
|
|
3503
|
+
|
|
3504
|
+
By the time either can fail, the status flip has landed and `save_state` has not — so
|
|
3505
|
+
the abort would otherwise leave the run's task ESCALATED against a spec already
|
|
3506
|
+
flipped to the re-drive's status and (for the re-stamp) stripped of the terminal
|
|
3507
|
+
`## Auto Run Result` the next resolve session reads as its context. That is exactly
|
|
3508
|
+
the "one edit nothing else records" the sequencing exists to prevent.
|
|
3509
|
+
|
|
3510
|
+
Writes only what it can prove it changed. `original` is `None` when the spec was
|
|
3511
|
+
unreadable before the first write (there is then nothing to restore, and nothing
|
|
3512
|
+
could have been written either), and a spec that is gone or unreadable NOW is not a
|
|
3513
|
+
state this undo can improve — recreating a file another process removed would fight
|
|
3514
|
+
a concurrent actor rather than restore this function's own edit. Bytes equal to
|
|
3515
|
+
`original` mean nothing landed, so nothing is rewritten and the mtime is left alone.
|
|
3516
|
+
|
|
3517
|
+
Byte-verbatim and CONFINED, matching the writes it undoes: `atomic_write_text_confined`
|
|
3518
|
+
would re-encode and translate newlines, so a CRLF spec would come back subtly
|
|
3519
|
+
different from the file this re-arm found, and an unconfined write would drop the
|
|
3520
|
+
`O_NOFOLLOW` walk of the parent components (#593) that every other write to this path
|
|
3521
|
+
takes. A restore that itself fails RAISES rather than degrading — the spec is then
|
|
3522
|
+
half-written and only the operator can settle it, which is the loudest thing this can
|
|
3523
|
+
be. `UnconfinedWriteError` is an `OSError`, so the one arm covers both.
|
|
3524
|
+
"""
|
|
3525
|
+
if original is None:
|
|
3526
|
+
return
|
|
3527
|
+
try:
|
|
3528
|
+
if spec_path.read_bytes() == original:
|
|
3529
|
+
return
|
|
3530
|
+
except OSError:
|
|
3531
|
+
return
|
|
3532
|
+
try:
|
|
3533
|
+
atomic_write_bytes_confined(spec_path, original, confine_root=task_spec_root(task, state))
|
|
3534
|
+
except OSError as e:
|
|
3535
|
+
raise RearmError(
|
|
3536
|
+
f"cannot restore {spec_path} after a failed re-arm "
|
|
3537
|
+
f"({e.__class__.__name__}: {e}) — the spec carries this re-arm's status flip "
|
|
3538
|
+
"and has lost its `## Auto Run Result` section, while the story is still "
|
|
3539
|
+
"escalated; restore the spec from git, then re-run resolve"
|
|
3540
|
+
) from e
|
|
3541
|
+
|
|
3542
|
+
|
|
3543
|
+
def _redrive_spec_status(state: RunState, task: StoryTask, *, isolated_redrive: bool) -> str:
|
|
3544
|
+
"""The spec's status AS THE RE-DRIVE WILL READ IT, or ``""`` when unprovable.
|
|
3545
|
+
|
|
3546
|
+
The proof that decides whether the operator still has anything to do, so it has to
|
|
3547
|
+
read the same file the caller's remedy names — otherwise the record holds a resume
|
|
3548
|
+
over work that is already done, or clears on work that is not.
|
|
3549
|
+
|
|
3550
|
+
Two sources, because the two re-drive modes read two different things:
|
|
3551
|
+
|
|
3552
|
+
* MOUNTING: the fresh worktree is cut from git and checks out TRACKED files only, so
|
|
3553
|
+
it reads the COMMITTED spec and never a working-tree write. Anchored on
|
|
3554
|
+
`state.code_root` — the same tree the baseline advance reads — at the ref
|
|
3555
|
+
`redrive_base_ref` names, the run's pinned `target_branch` rather than that tree's
|
|
3556
|
+
current `HEAD`.
|
|
3557
|
+
* IN PLACE: the story re-runs in the main checkout, which reads its WORKING TREE. A
|
|
3558
|
+
commit is neither required nor sufficient there, so measuring the committed tree
|
|
3559
|
+
would hold the resume until the operator committed a correction the re-drive would
|
|
3560
|
+
have read uncommitted — and `rearm_event_notice`'s in-place remedy tells them to
|
|
3561
|
+
do exactly that (re-apply it in the main checkout, no commit), so a committed-only
|
|
3562
|
+
proof would make the record's own instruction unable to clear it.
|
|
3563
|
+
|
|
3564
|
+
Reached only when the write does NOT reach the re-drive, so the in-place arm is
|
|
3565
|
+
always the isolation-flip shape: the flip's write landed in the mount the escalated
|
|
3566
|
+
attempt recorded while the re-drive reads `state.project`. That is the tree
|
|
3567
|
+
`task_spec_root` answers for a task with no mount, which is what the resume makes
|
|
3568
|
+
this task once `release_mount_owned_state` runs.
|
|
3569
|
+
|
|
3570
|
+
Degrades to ``""`` on every uncertainty: a spec recorded absolute (nothing names
|
|
3571
|
+
its position in the tree), an absent or non-blob path at that ref, a non-UTF-8 blob,
|
|
3572
|
+
or any `GitError` — which includes the project simply not being a repository, and a
|
|
3573
|
+
`target_branch` the code root no longer carries. ``""`` never equals a target
|
|
3574
|
+
status, so the caller's record still fires. Suppression therefore requires PROOF
|
|
3575
|
+
that the work is already done, and the non-repo case stays non-fatal, as the story's
|
|
3576
|
+
Boundaries require.
|
|
3577
|
+
|
|
3578
|
+
Degrades to ``""`` on every uncertainty in BOTH arms, including a spec recorded
|
|
3579
|
+
absolute. That arm is narrower than it looks: the caller has already answered the one
|
|
3580
|
+
absolute shape whose write the re-drive DOES read — the shared external spec — with
|
|
3581
|
+
`_spec_is_shared_with_the_redrive`. What still reaches here is an absolute spelling
|
|
3582
|
+
of a path inside one of the two checkouts, which is genuinely unreachable, and whose
|
|
3583
|
+
position in the re-drive's tree nothing here can name, so degrading it to a warning
|
|
3584
|
+
is the right answer rather than a gap.
|
|
3585
|
+
"""
|
|
3586
|
+
raw = Path(task.spec_file or "")
|
|
3587
|
+
if not task.spec_file or raw.is_absolute():
|
|
3588
|
+
return ""
|
|
3589
|
+
if not isolated_redrive:
|
|
3590
|
+
try:
|
|
3591
|
+
text = (Path(state.project) / raw).read_text(encoding="utf-8")
|
|
3592
|
+
except (OSError, UnicodeDecodeError):
|
|
3593
|
+
return ""
|
|
3594
|
+
return status_of(parse_frontmatter(text))
|
|
3595
|
+
try:
|
|
3596
|
+
blob = verify.file_bytes_at_revision(
|
|
3597
|
+
state.code_root,
|
|
3598
|
+
redrive_base_ref(state, isolated_redrive=isolated_redrive),
|
|
3599
|
+
raw.as_posix(),
|
|
3600
|
+
)
|
|
3601
|
+
except verify.GitError:
|
|
3602
|
+
return ""
|
|
3603
|
+
if blob is None:
|
|
3604
|
+
return ""
|
|
3605
|
+
try:
|
|
3606
|
+
text = blob.decode("utf-8")
|
|
3607
|
+
except UnicodeDecodeError:
|
|
3608
|
+
return ""
|
|
3609
|
+
return status_of(parse_frontmatter(text))
|
|
3610
|
+
|
|
3611
|
+
|
|
3612
|
+
def restamp_code_root(run_dir: Path, repo_root: Path) -> str | None:
|
|
3613
|
+
"""Re-point a paused run's persisted code-root mirror at `repo_root` — the tree
|
|
3614
|
+
the caller is about to act in — and return the warning an operator must see when
|
|
3615
|
+
that MOVED a root the run had recorded (`None` when it already agreed, or when the
|
|
3616
|
+
run predates the field).
|
|
3617
|
+
|
|
3618
|
+
Exists because `rearm_escalation` reads that mirror OUT OF PROCESS
|
|
3619
|
+
(`RunState.code_root`) and has no `ProjectPaths` to consult, while `repo_root:` is
|
|
3620
|
+
re-read from config.yaml by every process that arms an engine. `cli._resume_paused_run`
|
|
3621
|
+
folds the same re-stamp into the one `save_state` that also carries the policy
|
|
3622
|
+
snapshot and the config digest — this is the seam for the surfaces that re-arm
|
|
3623
|
+
BEFORE they resume (`cli.cmd_resolve`, `TuiApp._do_rearm`), where that write lands
|
|
3624
|
+
too late to aim the re-arm.
|
|
3625
|
+
|
|
3626
|
+
The compare is exact and uncanonicalized, matching resume's: both sides are
|
|
3627
|
+
`str(paths.repo_root)` off `froidconfig.load_paths`, which resolves every member or
|
|
3628
|
+
raises, so they are spelled the same way whenever they name the same tree. An empty
|
|
3629
|
+
recorded root is a MISSING value, not a divergent one — a state.json written before
|
|
3630
|
+
the field existed — so it is migrated silently and reported as no move.
|
|
3631
|
+
|
|
3632
|
+
The message names neither tree, like resume's: what an operator needs is that the
|
|
3633
|
+
run has changed repositories, and the paths are the half that would put an
|
|
3634
|
+
attacker-controlled string on their terminal.
|
|
3635
|
+
"""
|
|
3636
|
+
state = load_state(run_dir)
|
|
3637
|
+
new = str(repo_root)
|
|
3638
|
+
if state.repo_root == new:
|
|
3639
|
+
return None
|
|
3640
|
+
moved = bool(state.repo_root)
|
|
3641
|
+
state.repo_root = new
|
|
3642
|
+
save_state(run_dir, state)
|
|
3643
|
+
if not moved:
|
|
3644
|
+
return None
|
|
3645
|
+
return (
|
|
3646
|
+
f"run {run_dir.name}: the code root in _froid/bmm/config.yaml has changed since "
|
|
3647
|
+
"this run started — the re-drive works in the tree configured now, while the "
|
|
3648
|
+
"baselines, preserve refs and branches this run already recorded name objects "
|
|
3649
|
+
"in the previous one. Restore the previous `repo_root:` value if you did not "
|
|
3650
|
+
"intend the move."
|
|
3651
|
+
)
|
|
3652
|
+
|
|
3653
|
+
|
|
3654
|
+
def rearm_escalation(
|
|
3655
|
+
run_dir: Path,
|
|
3656
|
+
story_key: str | None = None,
|
|
3657
|
+
*,
|
|
3658
|
+
restore_patch: str | None = None,
|
|
3659
|
+
isolated_redrive: bool,
|
|
3660
|
+
) -> str:
|
|
3661
|
+
"""Re-arm an escalation-paused story so the next resume re-drives it.
|
|
3662
|
+
|
|
3663
|
+
Flips the escalated task out of its terminal ESCALATED phase back to
|
|
3664
|
+
PENDING — which makes `_finish_inflight` reset the tree to the story's
|
|
3665
|
+
baseline and re-run it (clean rebuild) against the now-corrected frozen
|
|
3666
|
+
spec. The baseline itself is advanced to the CODE TREE's current HEAD
|
|
3667
|
+
(`state.code_root`, which is `paths.repo_root` — the tree the dev writer
|
|
3668
|
+
stamps from and the proof-of-work gate measures, and the same directory as
|
|
3669
|
+
`state.project` in every configuration without a `repo_root:` override) and
|
|
3670
|
+
the untracked snapshot refreshed, so commits and files the resolve session
|
|
3671
|
+
produced count as the rebuild's starting point, not as attempt debris to
|
|
3672
|
+
roll back. Strips the escalated attempt's stale `## Auto Run Result`
|
|
3673
|
+
section so the re-drive cannot read as terminal from its first save, and
|
|
3674
|
+
sets the spec's frontmatter status so step-01 routes to the right stage.
|
|
3675
|
+
Does NOT clear the pause; the caller resumes the run separately.
|
|
3676
|
+
|
|
3677
|
+
Two consequences of the reset are load-bearing and easy to undo by accident:
|
|
3678
|
+
|
|
3679
|
+
- `task.generation` is bumped, because `attempt` returning to 0 would
|
|
3680
|
+
otherwise let the re-drive re-mint a session id byte-equal to one the
|
|
3681
|
+
abandoned attempt already recorded (#705). `task.sessions` is deliberately
|
|
3682
|
+
NOT cleared — a second resolve cycle reads that run-dir audit trail — so
|
|
3683
|
+
the id is what has to change.
|
|
3684
|
+
- The spec's `baseline_revision` is re-stamped on BOTH legs, and only when the
|
|
3685
|
+
advance above actually RAN — `advanced` records that both git reads succeeded,
|
|
3686
|
+
not that HEAD changed, so a resolve session that committed nothing still
|
|
3687
|
+
re-stamps (with the same sha, harmlessly). What it will not do is re-stamp
|
|
3688
|
+
after a FAILED advance (see the block that does it for why each half of that
|
|
3689
|
+
is the way it is).
|
|
3690
|
+
|
|
3691
|
+
Two re-drive modes, selected by `restore_patch`:
|
|
3692
|
+
|
|
3693
|
+
- **from-scratch** (default, ``restore_patch=None``): status → ``ready-for-dev``
|
|
3694
|
+
so the dev session re-implements from a clean baseline. Assigning None also
|
|
3695
|
+
clears any stale latch from a prior restore attempt the human abandoned.
|
|
3696
|
+
- **patch-restore** (Froid Plane #2564, ``restore_patch`` set): the human
|
|
3697
|
+
confirmed the escalated attempt's reading was correct. Status → ``in-review``
|
|
3698
|
+
so step-01 routes straight to step-04, and the path is latched onto the task
|
|
3699
|
+
(`task.restore_patch`) so the engine re-applies the saved patch onto the
|
|
3700
|
+
baseline before dispatching — the re-driven session resumes review on the
|
|
3701
|
+
restored diff instead of re-implementing. The status is set here
|
|
3702
|
+
deterministically; the resolve agent must NOT set it. Because the baseline
|
|
3703
|
+
advances (above) while the patch was diffed from the OLD baseline, a resolve
|
|
3704
|
+
session that committed changes to the patched files makes the re-drive's
|
|
3705
|
+
apply fail — the engine then escalates loudly instead of dispatching on a
|
|
3706
|
+
half-restored tree (see verify.apply_patch).
|
|
3707
|
+
|
|
3708
|
+
Stories mode: when the escalated spec is a fixed-slug sentinel
|
|
3709
|
+
(`<id>-unresolved.md` / `<id>-ambiguous.md`, written by a pre-planning HALT),
|
|
3710
|
+
it cannot be re-opened by a status flip — its very presence wedges the id.
|
|
3711
|
+
Instead preserve a copy under `{run_dir}/sentinels/`, journal `sentinel-cleared`
|
|
3712
|
+
with the blocking condition, and delete it, so the re-dispatch resolves to a
|
|
3713
|
+
clean PENDING and re-plans from scratch (leg 1 again for a spec_checkpoint id).
|
|
3714
|
+
|
|
3715
|
+
`isolated_redrive` is the LIVE policy's isolation mode (`scm.isolation ==
|
|
3716
|
+
"worktree"`), which run state cannot carry: the mode is re-read at every resume and
|
|
3717
|
+
a mid-run change is journalled, never refused, so the recorded `task.worktree_path`
|
|
3718
|
+
says how the escalated attempt RAN and only policy says how the re-drive WILL run.
|
|
3719
|
+
Keyword-only and required, because every consumer of it here is an answer a human
|
|
3720
|
+
acts on — which ref to commit the corrected spec on, whether the working-tree flip
|
|
3721
|
+
reaches the re-drive at all, whether a restore latch can be honored — and a
|
|
3722
|
+
defaulted mode would answer all three for the wrong tree in silence, which is the
|
|
3723
|
+
defect this parameter exists to close. Both callers (`cli.cmd_resolve`,
|
|
3724
|
+
`tui.TuiApp._do_rearm`) hold a loaded policy already.
|
|
3725
|
+
|
|
3726
|
+
Returns the re-armed story key. Raises RearmError when the run is not paused at
|
|
3727
|
+
the escalation stage, the target story is not escalated, or a supplied
|
|
3728
|
+
`restore_patch` fails `validate_restore_latch` (the shared precondition set —
|
|
3729
|
+
sentinel wedge, spec-less escalation, worktree isolation).
|
|
3730
|
+
"""
|
|
3731
|
+
state = load_state(run_dir)
|
|
3732
|
+
if state.paused_stage != PAUSE_ESCALATION:
|
|
3733
|
+
raise RearmError(
|
|
3734
|
+
f"run {run_dir.name} is not paused at an escalation "
|
|
3735
|
+
f"(stage: {state.paused_stage or 'none'})"
|
|
3736
|
+
)
|
|
3737
|
+
key = story_key or state.paused_story_key
|
|
3738
|
+
if key is None:
|
|
3739
|
+
raise RearmError(f"run {run_dir.name} has no escalated story to resolve")
|
|
3740
|
+
task = state.tasks.get(key)
|
|
3741
|
+
if task is None:
|
|
3742
|
+
raise RearmError(f"run {run_dir.name} has no task for story {key}")
|
|
3743
|
+
if task.phase != Phase.ESCALATED:
|
|
3744
|
+
raise RearmError(f"story {key} is not escalated (phase: {task.phase})")
|
|
3745
|
+
# Patch-restore preconditions (T1 guard + spec-less wedge + worktree isolation),
|
|
3746
|
+
# rejected here before any task mutation so the escalation stays armed for a
|
|
3747
|
+
# corrected resolve. `cli._resolve_restore_patch` runs the same validator ahead
|
|
3748
|
+
# of the interactive session; this call is what makes a programmatic caller
|
|
3749
|
+
# (TUI restore parity, scripts) unable to bypass it.
|
|
3750
|
+
if restore_patch:
|
|
3751
|
+
err = validate_restore_latch(state, task, key, worktree_isolation=isolated_redrive)
|
|
3752
|
+
if err is not None:
|
|
3753
|
+
raise RearmError(err)
|
|
3754
|
+
|
|
3755
|
+
journal = Journal(run_dir)
|
|
3756
|
+
# Read before the unconditional overwrite below: they describe the restore
|
|
3757
|
+
# attempt this re-arm is abandoning, and the residue block needs both.
|
|
3758
|
+
old_latch = task.restore_patch
|
|
3759
|
+
old_baseline = task.baseline_commit
|
|
3760
|
+
# deliberate reset, not a normal state-machine transition (mirrors
|
|
3761
|
+
# engine._finish_inflight): a clean re-attempt against the corrected spec.
|
|
3762
|
+
task.phase = Phase.PENDING
|
|
3763
|
+
task.attempt = 0
|
|
3764
|
+
# A new generation of this task. `attempt` going back to 0 (and the next
|
|
3765
|
+
# dispatch bumping it to 1) would otherwise re-mint a session task_id
|
|
3766
|
+
# byte-equal to one the abandoned attempt already recorded, and
|
|
3767
|
+
# `Engine._resumable_session` — matching that id over the append-only
|
|
3768
|
+
# `task.sessions`, which this function deliberately does NOT clear — would
|
|
3769
|
+
# replay the abandoned verdict for the fresh attempt (#705). Bumped BEFORE any
|
|
3770
|
+
# dispatch, so the id is unique from the re-drive's first session onward.
|
|
3771
|
+
task.generation += 1
|
|
3772
|
+
task.review_cycle = 0
|
|
3773
|
+
task.followup_reviews_spent = 0 # human-resolved re-drive gets a fresh damping budget
|
|
3774
|
+
task.defer_reason = None
|
|
3775
|
+
task.rearmed = True # resume-time recovery notice describes a clean rebuild,
|
|
3776
|
+
# not a failed attempt (engine._finish_inflight clears it once the rebuild runs)
|
|
3777
|
+
# Always (re)assign the latch: a None restore_patch clears a stale one left by
|
|
3778
|
+
# a prior restore attempt the human then chose to redo from scratch.
|
|
3779
|
+
task.restore_patch = restore_patch
|
|
3780
|
+
|
|
3781
|
+
# The bytes this re-arm found on the spec, for `_restore_rearmed_spec`. Declared out
|
|
3782
|
+
# here because the baseline re-stamp that consumes it sits in a SECOND
|
|
3783
|
+
# `if task.spec_file:` block, past the advance it depends on.
|
|
3784
|
+
spec_before: bytes | None = None
|
|
3785
|
+
if task.spec_file:
|
|
3786
|
+
spec_path = task_spec_path(task, state)
|
|
3787
|
+
# Stories mode only: a fixed-slug pre-planning-halt sentinel
|
|
3788
|
+
# (`<id>-unresolved.md` / `<id>-ambiguous.md`) is cleared by deletion, not a
|
|
3789
|
+
# status flip. Clear it ONLY when the run recorded this task AS a sentinel at
|
|
3790
|
+
# detection time (`task.sentinel_kind`, stamped by StoriesEngine's pick-time
|
|
3791
|
+
# wedge / post-dev read-back) — never by re-deriving from the basename. That
|
|
3792
|
+
# keeps a real story spec that merely happens to be named `<key>-unresolved.md`,
|
|
3793
|
+
# or a *non-sentinel* escalation whose spec matches the convention, on the
|
|
3794
|
+
# status-flip path so it is kept, not deleted. Gate on the run source too (the
|
|
3795
|
+
# convention exists only in stories mode) and defensively re-confirm the
|
|
3796
|
+
# on-disk name still matches the recorded slug before deleting.
|
|
3797
|
+
sentinel_kind = task.sentinel_kind if state.source == "stories" else ""
|
|
3798
|
+
if sentinel_kind and _sentinel_condition(spec_path, key) == sentinel_kind:
|
|
3799
|
+
# a sentinel is cleared by deletion, not a status flip; drop the stale
|
|
3800
|
+
# spec_file so the re-dispatch starts from PENDING (clean re-plan).
|
|
3801
|
+
_clear_sentinel(run_dir, journal, spec_path, key, sentinel_kind)
|
|
3802
|
+
task.spec_file = None
|
|
3803
|
+
task.sentinel_kind = "" # verdict discharged; the re-dispatch is clean
|
|
3804
|
+
# Deleting the sentinel does not make the re-plan produce a different one:
|
|
3805
|
+
# the correction that does lives UPSTREAM, in the `SPEC.md` / `stories.yaml`
|
|
3806
|
+
# the resolve skill sends the agent to instead of this file. That correction
|
|
3807
|
+
# faces the same reachability gap the spec arm below measures, and faced NO
|
|
3808
|
+
# gate at all — this arm cleared `spec_file` and fell through, so
|
|
3809
|
+
# `write_reaches_the_redrive` was never computed and the resume was never
|
|
3810
|
+
# held for a sentinel. An isolated re-drive then mounts fresh from
|
|
3811
|
+
# `redrive_base_ref`, re-plans from a committed tree that never saw the
|
|
3812
|
+
# edit, mints the same sentinel again, and the escalation is spent.
|
|
3813
|
+
#
|
|
3814
|
+
# Narrowed by PROOF for the reason the spec record below is, and the need is
|
|
3815
|
+
# sharper here: `stories_reach_the_redrive` answers "unreachable" for EVERY
|
|
3816
|
+
# isolated stories run whose spec folder sits inside the project, which is
|
|
3817
|
+
# every one we author. Gating on it alone would fire — and hold the resume —
|
|
3818
|
+
# on 100% of isolated sentinel re-arms, a per-configuration constant rather
|
|
3819
|
+
# than an event. `_redrive_reads_the_upstream_artifacts` is what makes it an
|
|
3820
|
+
# event: it fires only while this checkout still holds upstream bytes the
|
|
3821
|
+
# ref the re-drive mounts from does not.
|
|
3822
|
+
#
|
|
3823
|
+
# No `redrive` discriminator, unlike the spec record: this one has a single
|
|
3824
|
+
# remedy because it has a single reachable shape. An in-place re-drive reads
|
|
3825
|
+
# the main checkout's working tree, which is exactly where `cwd=project` put
|
|
3826
|
+
# the correction, so `stories_reach_the_redrive` short-circuits that leg to
|
|
3827
|
+
# reachable and no record is written for it at all.
|
|
3828
|
+
if not stories_reach_the_redrive(
|
|
3829
|
+
task, state, isolated_redrive=isolated_redrive
|
|
3830
|
+
) and not _redrive_reads_the_upstream_artifacts(state):
|
|
3831
|
+
journal.append(
|
|
3832
|
+
"rearm-upstream-write-unreachable",
|
|
3833
|
+
story_key=key,
|
|
3834
|
+
# `task_stories_root` names the tree the RUN owns; the correction
|
|
3835
|
+
# lands in the checkout the resolve session ran in. Both are the
|
|
3836
|
+
# project on this leg unless a mount is recorded, and the operator
|
|
3837
|
+
# needs the folder to act, so the record carries the folder the
|
|
3838
|
+
# remedy is about rather than the run's read locator.
|
|
3839
|
+
stories_root=str(_upstream_artifacts_folder(state)),
|
|
3840
|
+
target_branch=state.target_branch,
|
|
3841
|
+
)
|
|
3842
|
+
else:
|
|
3843
|
+
# A WORKTREE-LOCAL spec's writes below land in the unit's worktree
|
|
3844
|
+
# (`task_spec_path`) — which the re-drive destroys before reading anything.
|
|
3845
|
+
# A re-armed task (phase PENDING, `defer_reason` cleared, and no resumable
|
|
3846
|
+
# session because `generation` was just bumped) falls to
|
|
3847
|
+
# `engine._finish_inflight`'s final arm, which calls `discard_worktree` and
|
|
3848
|
+
# lets `_run_story` mount a fresh one. The re-driven session then resolves
|
|
3849
|
+
# its spec through `verify.resolve_spec_path(task.spec_file,
|
|
3850
|
+
# workspace.paths)` (`engine._dispatched_spec_for_attempt`), and under
|
|
3851
|
+
# isolation `workspace.paths` is rebased onto that FRESH worktree, which
|
|
3852
|
+
# checks out TRACKED files only. So the re-drive reads the COMMITTED spec.
|
|
3853
|
+
#
|
|
3854
|
+
# No working-tree write reaches it — not this one, and not a write to the
|
|
3855
|
+
# main checkout either: the fresh worktree comes from git rather than from a
|
|
3856
|
+
# copy of that tree, and `seed_adapter_defaults` seeds adapter config files,
|
|
3857
|
+
# not the output folder. The channel that DOES work is the human committing
|
|
3858
|
+
# the corrected spec from the resolve session, which runs with `cwd=project`.
|
|
3859
|
+
# The writes below are kept (they are correct for the in-place case, and
|
|
3860
|
+
# harmless here), but the operator is told — a flip that cannot land is
|
|
3861
|
+
# exactly the silent re-wedge #640(b) exists to end.
|
|
3862
|
+
#
|
|
3863
|
+
# "Worktree-local" is the load-bearing qualifier, and isolation does not
|
|
3864
|
+
# imply it: an artifact dir configured OUTSIDE the project tree is shared
|
|
3865
|
+
# across checkouts by `ProjectPaths.rebased`, so a spec that landed there is
|
|
3866
|
+
# one file the fresh worktree reads through the very absolute path this
|
|
3867
|
+
# writes to. `_spec_is_shared_with_the_redrive` carves out that case, and only
|
|
3868
|
+
# that one: the main checkout's copy is outside the worktree too, and stays
|
|
3869
|
+
# unreachable because the re-drive measures it against worktree-local roots.
|
|
3870
|
+
# Route /froid-build-auto via the spec's frontmatter status (decision
|
|
3871
|
+
# table): patch-restore -> in-review -> step-04 (resume review on
|
|
3872
|
+
# the restored diff); from-scratch -> ready-for-dev -> step-03
|
|
3873
|
+
# (re-implement). Independent of the resolve agent having set it.
|
|
3874
|
+
target_status = "in-review" if restore_patch else "ready-for-dev"
|
|
3875
|
+
# Whether the writes below are the copy the re-driven session actually
|
|
3876
|
+
# reads. Hoisted out of the record's condition because TWO decisions turn on
|
|
3877
|
+
# it, and only one of them used to: the warning below, and the flip's
|
|
3878
|
+
# REFUSAL one screen down, which was gated on `spec_path.is_file()` alone.
|
|
3879
|
+
# Under isolation that readable file is the doomed worktree copy, so the
|
|
3880
|
+
# refusal demanded a repair to the one file the re-drive destroys before
|
|
3881
|
+
# reading anything — and demanded it even when `_redrive_spec_status` had
|
|
3882
|
+
# already proven the committed spec carries the status the re-drive routes
|
|
3883
|
+
# on. See `_spec_is_shared_with_the_redrive` for why an isolated unit's spec
|
|
3884
|
+
# is nevertheless reachable when it sits in an artifact dir configured
|
|
3885
|
+
# outside the project tree.
|
|
3886
|
+
write_reaches_the_redrive = spec_reaches_the_redrive(
|
|
3887
|
+
task, state, isolated_redrive=isolated_redrive
|
|
3888
|
+
)
|
|
3889
|
+
# Narrowed to the case an operator can ACT on. Every isolated escalation
|
|
3890
|
+
# carries a mounted `worktree_path` — `worktree_flow.escalate_unit` never
|
|
3891
|
+
# clears it, and `keep_branch_and_escalate` deliberately leaves the worktree
|
|
3892
|
+
# up — so gating on that alone fired this warning on 100% of re-arms under
|
|
3893
|
+
# `isolation = "worktree"`: a per-configuration constant, not an event, and
|
|
3894
|
+
# the same "trains the operator to scroll past the meaningful one" failure
|
|
3895
|
+
# that the `flipped` read-back below and the `overwritten != old_baseline`
|
|
3896
|
+
# guard were each narrowed to avoid. The remedy it prints ("commit the
|
|
3897
|
+
# corrected spec") is already a no-op once the committed spec carries the
|
|
3898
|
+
# target status, which is precisely when the re-drive reads what it needs.
|
|
3899
|
+
# Suppression requires PROOF: an unreadable blob, a non-repo project, or any
|
|
3900
|
+
# git fault leaves `""` and the record fires. The proof is read at
|
|
3901
|
+
# `redrive_base_ref`, NOT at the code root's current `HEAD` — the two part
|
|
3902
|
+
# company as soon as the operator checks out another branch while the
|
|
3903
|
+
# escalation is paused, and this record now holds the resume.
|
|
3904
|
+
#
|
|
3905
|
+
# The branch rides along because the remedy needs it: on exactly the shape
|
|
3906
|
+
# the ref fix rescues, "commit the corrected spec" without a branch sends
|
|
3907
|
+
# the operator to commit again on the branch the re-drive does not read, and
|
|
3908
|
+
# the next re-arm prints the same sentence. Empty for the migrated shape
|
|
3909
|
+
# `redrive_base_ref` degrades to `HEAD` for, and the notice drops the
|
|
3910
|
+
# clause rather than naming a ref it cannot source — and empty for an
|
|
3911
|
+
# IN-PLACE re-drive, which has no branch to name at all.
|
|
3912
|
+
#
|
|
3913
|
+
# `redrive` is that second shape's discriminator, and it goes ON the record
|
|
3914
|
+
# because the reader is out of process: `rearm_event_notice` renders from a
|
|
3915
|
+
# journal line alone and cannot re-read the policy that produced it. One
|
|
3916
|
+
# kind, two remedies. Isolated: the writes landed in a mount the re-drive
|
|
3917
|
+
# discards, so the correction must be COMMITTED on the named branch. In
|
|
3918
|
+
# place: the writes landed in the mount the escalated attempt recorded while
|
|
3919
|
+
# the re-drive now reads the main checkout, so the correction must be made
|
|
3920
|
+
# THERE — a commit is neither required nor sufficient. Telling the second
|
|
3921
|
+
# operator to commit sends them to the wrong tree, which is the same class
|
|
3922
|
+
# of silent loss this whole record exists to end.
|
|
3923
|
+
#
|
|
3924
|
+
# Spelled `target_branch` and NOT `base`, because `diagnostics` routes the
|
|
3925
|
+
# scrub by field NAME: `target_branch` is already in `_JOURNAL_ALIAS_FIELDS`
|
|
3926
|
+
# under the `branch` namespace (with no journal producer until now), while
|
|
3927
|
+
# any new spelling falls through to `scrub_json`, which waves an
|
|
3928
|
+
# identifier-shaped branch name through verbatim. In a normal run
|
|
3929
|
+
# `ensure_target_branch` has already journalled the same string as `branch`,
|
|
3930
|
+
# so the egress backstop would repair it and disclose a `backstop_repairs`
|
|
3931
|
+
# routing gap; in a truncated journal missing that event nothing would catch
|
|
3932
|
+
# it and the branch would ship in a shareable bundle. `target` — the
|
|
3933
|
+
# spelling the merge kinds use — is NOT available: `board-advance-*` puts a
|
|
3934
|
+
# sprint STATUS in that same field, and routing is by name, so aliasing it
|
|
3935
|
+
# to `branch` would pseudonymize statuses as branches.
|
|
3936
|
+
if (
|
|
3937
|
+
not write_reaches_the_redrive
|
|
3938
|
+
and _redrive_spec_status(state, task, isolated_redrive=isolated_redrive)
|
|
3939
|
+
!= target_status
|
|
3940
|
+
):
|
|
3941
|
+
journal.append(
|
|
3942
|
+
"rearm-spec-write-unreachable",
|
|
3943
|
+
story_key=key,
|
|
3944
|
+
spec_file=str(spec_path),
|
|
3945
|
+
status=target_status,
|
|
3946
|
+
target_branch=state.target_branch if isolated_redrive else "",
|
|
3947
|
+
redrive="isolated" if isolated_redrive else "in-place",
|
|
3948
|
+
)
|
|
3949
|
+
# Captured immediately before the FIRST write, so an abort further down can
|
|
3950
|
+
# put the spec back exactly as found. Unreadable degrades to `None`: the
|
|
3951
|
+
# writes below answer such a path with `False` rather than an exception, so
|
|
3952
|
+
# there would be nothing to undo either.
|
|
3953
|
+
try:
|
|
3954
|
+
spec_before = spec_path.read_bytes()
|
|
3955
|
+
except OSError:
|
|
3956
|
+
spec_before = None
|
|
3957
|
+
try:
|
|
3958
|
+
flipped = verify.set_frontmatter_status(
|
|
3959
|
+
spec_path, target_status, confine_root=task_spec_root(task, state)
|
|
3960
|
+
)
|
|
3961
|
+
# `set_frontmatter_status` answers "nothing to change" with `False`
|
|
3962
|
+
# for FOUR causes, not three — its own docstring lists them: no file,
|
|
3963
|
+
# no frontmatter block, no top-level `status:`, and ALREADY AT THE
|
|
3964
|
+
# TARGET (`_edit_frontmatter_block` returns None on
|
|
3965
|
+
# `original[key] == value`). Only the first three are failures. The
|
|
3966
|
+
# fourth is an ordinary, fully-successful re-arm: a second resolve
|
|
3967
|
+
# cycle on an already-flipped spec, or the documented
|
|
3968
|
+
# `resolve --no-interactive` flow where a human fixed the spec
|
|
3969
|
+
# themselves — the case the comment above calls "Independent of the
|
|
3970
|
+
# resolve agent having set it". Journalling it fired the operator
|
|
3971
|
+
# warning ("could not be re-opened … may re-wedge on it") on a spec
|
|
3972
|
+
# that was byte-identical and CORRECT, which is the "trains the
|
|
3973
|
+
# operator to scroll past the meaningful one" failure the re-stamp's
|
|
3974
|
+
# `overwritten != old_baseline` guard exists to prevent one screen
|
|
3975
|
+
# below. Read the status back to tell the two apart: `read_frontmatter`
|
|
3976
|
+
# degrades a missing/unreadable/unparseable spec to `{}` and `status_of`
|
|
3977
|
+
# then answers `""`, so all three real failures still record.
|
|
3978
|
+
if not flipped and verify.status_of(verify.read_frontmatter(spec_path)) != (
|
|
3979
|
+
target_status
|
|
3980
|
+
):
|
|
3981
|
+
# Discarding that return is how the flip
|
|
3982
|
+
# became a SILENT no-op: the re-drive is dispatched anyway, step-01
|
|
3983
|
+
# reads the unchanged terminal status, routes the session to "ingest
|
|
3984
|
+
# as context, do not resume", and the story re-wedges with nothing on
|
|
3985
|
+
# the record. The `FrontmatterWriteError` arm below covers only the
|
|
3986
|
+
# shapes that RAISE; this covers the ones that lie quietly.
|
|
3987
|
+
# `refused` is written ON the record because ONE kind now covers
|
|
3988
|
+
# two outcomes and the operator surfaces must tell them apart —
|
|
3989
|
+
# they read the journal OUT OF PROCESS, with neither the task nor
|
|
3990
|
+
# the tree to re-derive it from. Printing the refusal's remedy
|
|
3991
|
+
# ("add a top-level `status:`") for a re-arm that COMPLETED sends
|
|
3992
|
+
# the human to repair a file nothing will read.
|
|
3993
|
+
refused = spec_path.is_file() and write_reaches_the_redrive
|
|
3994
|
+
journal.append(
|
|
3995
|
+
"rearm-spec-flip-skipped",
|
|
3996
|
+
story_key=key,
|
|
3997
|
+
spec_file=str(spec_path),
|
|
3998
|
+
status=target_status,
|
|
3999
|
+
refused=refused,
|
|
4000
|
+
)
|
|
4001
|
+
# ...and then ABORT — but only for a spec that IS a readable file
|
|
4002
|
+
# here AND is the copy the re-drive reads. The first half is the same
|
|
4003
|
+
# `is_file` split the baseline re-stamp below already draws, and for
|
|
4004
|
+
# the same reason. On THAT shape the failure is
|
|
4005
|
+
# a REPAIR that did not land on the very file the re-drive reads, so it
|
|
4006
|
+
# aborts for the same reason the `FrontmatterWriteError` arm does:
|
|
4007
|
+
# journalling alone left the two default surfaces telling the operator
|
|
4008
|
+
# "re-armed <story>" and resuming in the same gesture, so the record's
|
|
4009
|
+
# own imperative was already unactionable when it rendered — while
|
|
4010
|
+
# step-01's contract for what reaches here is not a maybe. A spec with
|
|
4011
|
+
# no `status:` HALTs blocked on `unrecognized status in existing story
|
|
4012
|
+
# file`; one still carrying the escalated attempt's terminal status
|
|
4013
|
+
# routes to "ingest as context, do not resume". Either way the re-drive
|
|
4014
|
+
# re-wedges and the escalation is burned. Refusing keeps it armed: nothing
|
|
4015
|
+
# is persisted yet (`save_state` runs below), the spec is byte-identical
|
|
4016
|
+
# (the `## Auto Run Result` strip is deliberately sequenced AFTER this
|
|
4017
|
+
# check so an abort leaves nothing half-done), and the human fixes the
|
|
4018
|
+
# frontmatter and re-runs resolve.
|
|
4019
|
+
#
|
|
4020
|
+
# A spec that is NOT a file from here keeps warn-and-continue, because
|
|
4021
|
+
# there the flip's failure says nothing about what the re-drive will
|
|
4022
|
+
# read: `spec_file` is persisted RELATIVE to a worktree, an isolated
|
|
4023
|
+
# task's worktree may already be gone, and the re-drive mounts a fresh
|
|
4024
|
+
# one and reads the COMMITTED spec regardless. Aborting on it would
|
|
4025
|
+
# refuse the re-arms that the `rearm-baseline-restamp-skipped` and
|
|
4026
|
+
# `rearm-spec-write-unreachable` records exist to report rather than
|
|
4027
|
+
# prevent — an unreadable path is an observation, and observations
|
|
4028
|
+
# degrade.
|
|
4029
|
+
#
|
|
4030
|
+
# A worktree-local spec that IS readable takes that same lane, for a
|
|
4031
|
+
# sharper version of the same reason: `task_spec_root` anchors this
|
|
4032
|
+
# write on the mounted worktree, so the readable file is the copy the
|
|
4033
|
+
# re-drive DISCARDS. The refusal's own remedy could not fix anything
|
|
4034
|
+
# there — an operator who added a `status:` to that file and re-ran
|
|
4035
|
+
# resolve would flip a spec that is deleted before it is read, while
|
|
4036
|
+
# the committed spec, the one thing that decides routing, went
|
|
4037
|
+
# untouched. Worse, the refusal fired even when the correction was
|
|
4038
|
+
# already committed: `_redrive_spec_status` had just PROVEN the
|
|
4039
|
+
# re-drive routes correctly, and the re-arm was refused anyway over an
|
|
4040
|
+
# obsolete copy. The real remedy on that shape is
|
|
4041
|
+
# `rearm-spec-write-unreachable`'s ("commit the corrected spec"),
|
|
4042
|
+
# which fires from the block above on exactly the legs that need it
|
|
4043
|
+
# and now holds the resume rather than merely printing.
|
|
4044
|
+
#
|
|
4045
|
+
# The record is written on BOTH sides of that split: the abort message
|
|
4046
|
+
# reaches stderr only, and the journal is the run's audit trail —
|
|
4047
|
+
# `_echo_rearm_events` surfaces it from a `finally` on this path.
|
|
4048
|
+
if refused:
|
|
4049
|
+
raise RearmError(
|
|
4050
|
+
f"cannot re-open story spec {spec_path} to `{target_status}` "
|
|
4051
|
+
"for the re-drive: it has no frontmatter `status:` this re-arm "
|
|
4052
|
+
"can set, so the re-driven session would wedge on the status "
|
|
4053
|
+
"it reads — add a top-level `status:` to the spec's "
|
|
4054
|
+
"frontmatter block, then re-run resolve"
|
|
4055
|
+
)
|
|
4056
|
+
# drop the stale `## Auto Run Result` section along with the status flip
|
|
4057
|
+
# (mirrors engine._reset_spec_for_repair): find_result_artifact keys on
|
|
4058
|
+
# that heading, so leaving it would let the re-driven session's first
|
|
4059
|
+
# save of the spec parse as the prior attempt's terminal outcome.
|
|
4060
|
+
#
|
|
4061
|
+
# Sequenced AFTER the read-back check above, not with the flip it mirrors:
|
|
4062
|
+
# that check now raises, and an aborted re-arm must leave the spec exactly
|
|
4063
|
+
# as it found it — a stripped result section on a spec the re-arm then
|
|
4064
|
+
# refused would be the one edit nothing else records.
|
|
4065
|
+
devcontract.strip_auto_run_result(
|
|
4066
|
+
spec_path, confine_root=task_spec_root(task, state)
|
|
4067
|
+
)
|
|
4068
|
+
except verify.FrontmatterWriteError as e:
|
|
4069
|
+
# The spec reads fine but carries `status:` in a shape no line
|
|
4070
|
+
# edit can move (a block scalar, a flow mapping, a value continued
|
|
4071
|
+
# on the next line). This used to be a silent no-op on a bool
|
|
4072
|
+
# nobody read: the re-drive was dispatched anyway, step-01 saw the
|
|
4073
|
+
# unchanged terminal status and routed the session to "ingest as
|
|
4074
|
+
# context, do not resume", and the story re-wedged with nothing on
|
|
4075
|
+
# the record explaining why. Abort here for the same reason as
|
|
4076
|
+
# below, with the remedy this cause actually has.
|
|
4077
|
+
raise RearmError(
|
|
4078
|
+
f"cannot re-open story spec {spec_path} for the re-drive: {e} "
|
|
4079
|
+
f"— the re-drive would repeat the wedge it is meant to clear"
|
|
4080
|
+
) from e
|
|
4081
|
+
except (OSError, UnicodeDecodeError) as e:
|
|
4082
|
+
# Both helpers re-read the spec as UTF-8; an undecodable PRESENT
|
|
4083
|
+
# spec is a first-class escalation state (resolve_story_spec
|
|
4084
|
+
# degrades it to a wedge), so it can reach this flip. Without the
|
|
4085
|
+
# flip the re-drive would just re-wedge — abort BEFORE any state
|
|
4086
|
+
# is persisted (save_state runs below) with an actionable error
|
|
4087
|
+
# instead of a traceback; the escalation stays armed for a retry.
|
|
4088
|
+
#
|
|
4089
|
+
# ...and this arm is the SECOND refusal that can fire after a write has
|
|
4090
|
+
# landed, which the sequencing argument above does not cover. It guards
|
|
4091
|
+
# BOTH helpers, and `strip_auto_run_result` is the later one: by the
|
|
4092
|
+
# time its own read/decode or its atomic write faults (an
|
|
4093
|
+
# `atomic_write_bytes_confined` that cannot land — ENOSPC, EIO, a
|
|
4094
|
+
# component swapped for a link under the `O_NOFOLLOW` walk — or a spec
|
|
4095
|
+
# replaced under us between the two writes), the flip has already been
|
|
4096
|
+
# published and `save_state` has not. Ordering the strip after the
|
|
4097
|
+
# read-back check bought that check its byte-identical abort; it buys
|
|
4098
|
+
# this one nothing, because the fault is IN the strip. So the same undo
|
|
4099
|
+
# the re-stamp carries applies here, on the same terms.
|
|
4100
|
+
#
|
|
4101
|
+
# On the arm's other shape — the flip itself faulting on an
|
|
4102
|
+
# unreadable/undecodable spec — nothing was written, `spec_before` still
|
|
4103
|
+
# equals the bytes on disk, and `_restore_rearmed_spec` proves that and
|
|
4104
|
+
# returns without touching the file or its mtime.
|
|
4105
|
+
_restore_rearmed_spec(spec_path, spec_before, task, state)
|
|
4106
|
+
raise RearmError(
|
|
4107
|
+
f"cannot re-open story spec {spec_path} for the re-drive "
|
|
4108
|
+
f"({e.__class__.__name__}: {e}) — fix or replace the file "
|
|
4109
|
+
f"(it must be readable UTF-8), then re-run resolve"
|
|
4110
|
+
) from e
|
|
4111
|
+
|
|
4112
|
+
# A previous restore latch is being replaced (or re-latched onto the same
|
|
4113
|
+
# patch): the abandoned attempt applied that patch, so its NEW files sit
|
|
4114
|
+
# untracked in the tree right now. The refresh below would capture them as
|
|
4115
|
+
# "pre-existing" — after which every rollback preserves them and
|
|
4116
|
+
# finalize_commit's `add -A` sweeps the abandoned attempt into the corrected
|
|
4117
|
+
# story's commit. Subtract them instead (issue #90).
|
|
4118
|
+
#
|
|
4119
|
+
# Runs after the spec block for the same reason the refresh does (a cleared
|
|
4120
|
+
# sentinel must not be snapshotted), and before it because it feeds it.
|
|
4121
|
+
# Nothing is deleted here: the re-drive's reset (verify.safe_rollback) removes
|
|
4122
|
+
# whatever the refreshed snapshot no longer blesses, at the right moment.
|
|
4123
|
+
# The CODE tree, not `state.project`: every git read below (and every baseline
|
|
4124
|
+
# the proof-of-work gate later measures against) must name the repository the
|
|
4125
|
+
# dev writer stamps.
|
|
4126
|
+
#
|
|
4127
|
+
# That is `paths.repo_root` for every run this function can be reached from, but
|
|
4128
|
+
# NOT because `paths.repo_root == workspace.root` universally — it does not.
|
|
4129
|
+
# `Workspace.default` sets `root=paths.repo_root`, while the isolation constructor
|
|
4130
|
+
# mounts `root=<run_dir>/worktrees/<unit>` and rebases a fresh `ProjectPaths` onto
|
|
4131
|
+
# it, so under `isolation = "worktree"` the run-level `repo_root` is the main
|
|
4132
|
+
# checkout and the baseline is stamped in the worktree.
|
|
4133
|
+
#
|
|
4134
|
+
# `froidconfig.worktree_isolation_conflict` refuses worktree isolation beside a
|
|
4135
|
+
# `repo_root:` OVERRIDE — a narrower fact than it looks. It forces
|
|
4136
|
+
# `repo_root == project`; it says nothing about `repo_root` vs `workspace.root`.
|
|
4137
|
+
# Under plain isolation with NO override those two still diverge and isolation is
|
|
4138
|
+
# ON, so "wherever the roots could diverge, isolation is off" is false, and a rule
|
|
4139
|
+
# built on it licenses treating `state.code_root` as the tree the dev writer
|
|
4140
|
+
# stamped — which under isolation it is not.
|
|
4141
|
+
#
|
|
4142
|
+
# What is true, and the only claim to carry forward: `repo_root == project` in
|
|
4143
|
+
# every reachable configuration, so reading HEAD here is right for the in-place
|
|
4144
|
+
# case; and under isolation this value is deliberately SUPERSEDED rather than
|
|
4145
|
+
# relied on — `engine._finish_inflight` discards the worktree and `_dev_phase`
|
|
4146
|
+
# re-stamps `task.baseline_commit` from the fresh worktree's HEAD before any gate
|
|
4147
|
+
# reads it. Do not carry an identity into new code; carry this argument.
|
|
4148
|
+
#
|
|
4149
|
+
# A pre-upgrade state.json with no recorded root degrades to `project` exactly as
|
|
4150
|
+
# before.
|
|
4151
|
+
repo = state.code_root
|
|
4152
|
+
stale_residue = _stale_restore_residue(repo, journal, key, old_latch, old_baseline)
|
|
4153
|
+
|
|
4154
|
+
# Advance the attempt baseline to the CODE TREE's current HEAD (`repo`, above)
|
|
4155
|
+
# and refresh the untracked snapshot: whatever the human-driven resolve session left on the
|
|
4156
|
+
# branch (a committed fixture, a corrected ledger, ...) is authorized input
|
|
4157
|
+
# for the re-drive, not failed-attempt debris. Without this, the re-drive's
|
|
4158
|
+
# reset-to-baseline in engine._rollback_or_pause parks the resolution
|
|
4159
|
+
# commits on an attempt-preserve ref and rebuilds against a tree that
|
|
4160
|
+
# contradicts the corrected spec — the re-driven dev session then hits the
|
|
4161
|
+
# very gap the human just resolved. Best-effort: on a git failure the old
|
|
4162
|
+
# baseline stands (the redrive rollback path tolerates a stale baseline; it
|
|
4163
|
+
# just loses this protection).
|
|
4164
|
+
# Runs AFTER the spec block so a just-cleared stories sentinel (an untracked
|
|
4165
|
+
# file removed above) is not captured into baseline_untracked as a phantom
|
|
4166
|
+
# pre-existing untracked file. The two locals are computed before either task
|
|
4167
|
+
# field is assigned, so a failure on either git call can't advance
|
|
4168
|
+
# baseline_commit while baseline_untracked stays stale, or vice versa.
|
|
4169
|
+
advanced = False
|
|
4170
|
+
try:
|
|
4171
|
+
head = verify.rev_parse_head(repo)
|
|
4172
|
+
untracked = sorted(verify.untracked_files(repo) - stale_residue)
|
|
4173
|
+
except verify.GitError as e:
|
|
4174
|
+
# `verify.GitError` is a TOTAL replacement for the `except Exception` that
|
|
4175
|
+
# stood here, not a narrowing that leaks: both calls go through `_run_git`,
|
|
4176
|
+
# which translates spawn (`GitSpawnError`), timeout (`GitTimeoutError`) and
|
|
4177
|
+
# decode faults into this one taxonomy, and a non-zero rc into a plain
|
|
4178
|
+
# `GitError`. Still swallowed rather than raised — a project that is not a
|
|
4179
|
+
# git repo must not fail re-arm — but no longer SILENT: the degrade is the
|
|
4180
|
+
# difference between "the re-drive starts from the resolution" and "it
|
|
4181
|
+
# rebuilds against the tree the human just corrected away", and the
|
|
4182
|
+
# re-stamp below now refuses to paper over it.
|
|
4183
|
+
journal.append(
|
|
4184
|
+
"rearm-baseline-advance-failed",
|
|
4185
|
+
story_key=key,
|
|
4186
|
+
repo=str(repo),
|
|
4187
|
+
baseline=old_baseline or "",
|
|
4188
|
+
error=f"{e.__class__.__name__}: {e}",
|
|
4189
|
+
)
|
|
4190
|
+
else:
|
|
4191
|
+
task.baseline_commit = head
|
|
4192
|
+
task.baseline_untracked = untracked
|
|
4193
|
+
advanced = True
|
|
4194
|
+
|
|
4195
|
+
# Re-stamp the spec's own baseline to the advanced one, on BOTH re-drive legs.
|
|
4196
|
+
#
|
|
4197
|
+
# The patch-restore leg needs it because the in-review route skips step-03 —
|
|
4198
|
+
# the only step that stamps `baseline_revision` — so without it the re-driven
|
|
4199
|
+
# step-04 would build its review diff (and, on an intent-gap/bad-spec
|
|
4200
|
+
# re-triage, revert) "since" the ORIGINAL pre-attempt sha, clawing back the
|
|
4201
|
+
# very resolve-session commits the advance above just blessed as the re-drive's
|
|
4202
|
+
# starting point.
|
|
4203
|
+
#
|
|
4204
|
+
# The from-scratch leg gets it too (#640a). Its step-03 re-stamps the key
|
|
4205
|
+
# itself, so the write is redundant on the happy path — but only ON that path:
|
|
4206
|
+
# until step-03 runs, the spec carries the escalated attempt's sha, and every
|
|
4207
|
+
# gate that reads a claimed baseline before then reads a stale one. The cost is
|
|
4208
|
+
# recorded rather than hidden: re-stamping removes the gate's INDEPENDENT
|
|
4209
|
+
# signal on this leg (it then compares a value the orchestrator itself wrote),
|
|
4210
|
+
# so a claim that genuinely diverged is journalled on the way out instead of
|
|
4211
|
+
# being silently normalized.
|
|
4212
|
+
#
|
|
4213
|
+
# Gated on `advanced`, not on truthiness of `task.baseline_commit`: a failed
|
|
4214
|
+
# advance leaves the OLD sha in that field, which passes a truthiness test
|
|
4215
|
+
# identically to a freshly advanced one. Writing it would make spec and task
|
|
4216
|
+
# agree on a stale value — the one state in which nothing downstream can tell
|
|
4217
|
+
# that the advance never happened, and the re-drive rebuilds from the wrong
|
|
4218
|
+
# point with no error anywhere. Skipping keeps the failure legible (the degrade
|
|
4219
|
+
# is journalled above) and keeps re-arm non-fatal outside a repo.
|
|
4220
|
+
#
|
|
4221
|
+
# Loud on WRITE failure: a silently stale spec baseline is exactly the hazard
|
|
4222
|
+
# being closed.
|
|
4223
|
+
#
|
|
4224
|
+
# Guarded on `is_file` FIRST, because a spec this process cannot reach is not a
|
|
4225
|
+
# write failure here — it is a SILENT one. Both frontmatter writers answer such a
|
|
4226
|
+
# path with `False` rather than an exception (`verify.set_frontmatter_status`,
|
|
4227
|
+
# `verify.set_frontmatter_field`), so without a check the re-stamp no-ops with
|
|
4228
|
+
# nothing on the record and the spec keeps the escalated attempt's sha.
|
|
4229
|
+
#
|
|
4230
|
+
# `task_spec_path` re-anchors the recorded path before we get here, which is what
|
|
4231
|
+
# makes `is_file` mean what it says. Resolved raw it meant something else and worse:
|
|
4232
|
+
# `spec_file` is persisted RELATIVE to the worktree for an isolated task, and the
|
|
4233
|
+
# main checkout carries the same layout, so the check passed on the wrong file and
|
|
4234
|
+
# the write landed there. The restore leg cannot reach any of this (its precondition
|
|
4235
|
+
# rejects a truthy `task.worktree_path`); the from-scratch leg has no such guard,
|
|
4236
|
+
# which is exactly why that precondition has to exist.
|
|
4237
|
+
#
|
|
4238
|
+
# `is_file` is necessary but not sufficient: a spec that EXISTS with no frontmatter
|
|
4239
|
+
# block also returns `False` from both writers. That shape is caught by the flip's
|
|
4240
|
+
# `flipped` check above and, here, by `overwritten` staying empty.
|
|
4241
|
+
if task.spec_file:
|
|
4242
|
+
spec_path = task_spec_path(task, state)
|
|
4243
|
+
if not spec_path.is_file():
|
|
4244
|
+
# OUTSIDE the `advanced` gate on purpose. Nesting this record inside it
|
|
4245
|
+
# made the two #640 legs shadow each other: on a project that is not a
|
|
4246
|
+
# repo the advance fails, `advanced` is False, and an unreadable spec
|
|
4247
|
+
# then produced NO record at all — the journal blamed git while the
|
|
4248
|
+
# status flip above had silently no-opped for an entirely different
|
|
4249
|
+
# reason. The two degrades compose; they do not substitute.
|
|
4250
|
+
journal.append(
|
|
4251
|
+
"rearm-baseline-restamp-skipped",
|
|
4252
|
+
story_key=key,
|
|
4253
|
+
spec_file=str(spec_path),
|
|
4254
|
+
baseline=task.baseline_commit or "",
|
|
4255
|
+
)
|
|
4256
|
+
elif advanced and task.baseline_commit:
|
|
4257
|
+
try:
|
|
4258
|
+
# Read through the same reader both consumers of a claimed baseline use,
|
|
4259
|
+
# so what gets journalled as "overwritten" is the value the gate would
|
|
4260
|
+
# have judged — not whichever key happened to be inspected here (#716).
|
|
4261
|
+
#
|
|
4262
|
+
# INSIDE the try, with the write it describes. `read_frontmatter` opens
|
|
4263
|
+
# the file itself, so an OSError here would otherwise escape as a
|
|
4264
|
+
# traceback from the one block whose whole contract is to turn a spec
|
|
4265
|
+
# this re-arm cannot move into an actionable `RearmError`. What it does
|
|
4266
|
+
# NOT rescue: `read_frontmatter` DEGRADES an unparseable YAML block to
|
|
4267
|
+
# `{}` rather than raising, so on such a spec `overwritten` is `""`, the
|
|
4268
|
+
# guard below is falsy, and no divergence record is written even though
|
|
4269
|
+
# the insert lands. That is the reader's deliberate observe-degrade
|
|
4270
|
+
# contract, not something to defeat here — the value is unknowable, and
|
|
4271
|
+
# inventing one would be worse than the silence.
|
|
4272
|
+
overwritten = auto_dev_baseline_of(verify.read_frontmatter(spec_path))
|
|
4273
|
+
verify.set_frontmatter_field(
|
|
4274
|
+
spec_path,
|
|
4275
|
+
"baseline_revision",
|
|
4276
|
+
task.baseline_commit,
|
|
4277
|
+
confine_root=task_spec_root(task, state),
|
|
4278
|
+
)
|
|
4279
|
+
except (OSError, UnicodeDecodeError, verify.FrontmatterWriteError) as e:
|
|
4280
|
+
# FrontmatterWriteError joins the tuple rather than getting its own
|
|
4281
|
+
# arm: the remedy is the same sentence ("fix the file"), and the
|
|
4282
|
+
# exception already says which shape it could not move. What matters
|
|
4283
|
+
# is that it aborts here — the stale-baseline hazard this block exists
|
|
4284
|
+
# to close is exactly what a swallowed write would leave behind.
|
|
4285
|
+
#
|
|
4286
|
+
# ...and that the abort leaves the spec as this re-arm FOUND it. This is
|
|
4287
|
+
# the LAST of the two refusals that can fire after a write has landed —
|
|
4288
|
+
# the flip and the result strip are both behind us, `save_state` is not —
|
|
4289
|
+
# so it carries the undo the sequenced refusals get for free (the other
|
|
4290
|
+
# is the spec block's `(OSError, UnicodeDecodeError)` arm, which the
|
|
4291
|
+
# strip raises through after the flip has published). Without
|
|
4292
|
+
# it a spec with a movable `status:` beside an unmovable
|
|
4293
|
+
# `baseline_revision:` came back flipped to the re-drive's status and
|
|
4294
|
+
# stripped of the terminal result, while the run still called the story
|
|
4295
|
+
# escalated.
|
|
4296
|
+
_restore_rearmed_spec(spec_path, spec_before, task, state)
|
|
4297
|
+
raise RearmError(
|
|
4298
|
+
f"cannot re-stamp baseline_revision on {spec_path} "
|
|
4299
|
+
f"({e.__class__.__name__}: {e}) — fix the file, then re-run resolve"
|
|
4300
|
+
) from e
|
|
4301
|
+
if overwritten and overwritten != old_baseline:
|
|
4302
|
+
# Compared against `old_baseline` — what the RUN recorded for the
|
|
4303
|
+
# escalated attempt — NOT against `task.baseline_commit`, which the
|
|
4304
|
+
# advance above has already moved to the new HEAD. Measuring against the
|
|
4305
|
+
# advanced value made this fire on every ordinary from-scratch re-arm
|
|
4306
|
+
# whose resolve session committed anything: the spec and the run agreed
|
|
4307
|
+
# exactly, and the operator was still told they diverged. A record that
|
|
4308
|
+
# fires on the routine case is the "trains the operator to scroll past
|
|
4309
|
+
# the meaningful one" failure the `restore` split exists to prevent.
|
|
4310
|
+
#
|
|
4311
|
+
# What survives is the real signal, on BOTH legs: the spec claimed a
|
|
4312
|
+
# baseline the run never recorded. That is the only trace left of a
|
|
4313
|
+
# divergence the gate can no longer report, because the re-stamp is
|
|
4314
|
+
# about to normalize it away.
|
|
4315
|
+
journal.append(
|
|
4316
|
+
"rearm-baseline-restamped",
|
|
4317
|
+
story_key=key,
|
|
4318
|
+
spec_file=str(spec_path),
|
|
4319
|
+
overwritten=overwritten,
|
|
4320
|
+
baseline=task.baseline_commit,
|
|
4321
|
+
restore=bool(restore_patch),
|
|
4322
|
+
)
|
|
4323
|
+
|
|
4324
|
+
save_state(run_dir, state)
|
|
4325
|
+
journal.append(
|
|
4326
|
+
"story-escalation-resolved",
|
|
4327
|
+
story_key=key,
|
|
4328
|
+
baseline=task.baseline_commit or "",
|
|
4329
|
+
restore=bool(restore_patch),
|
|
4330
|
+
)
|
|
4331
|
+
return key
|
|
4332
|
+
|
|
4333
|
+
|
|
4334
|
+
def journal_entries_or_none(run_dir: Path) -> list[dict[str, Any]] | None:
|
|
4335
|
+
"""This run's journal entries, or ``None`` when the journal cannot be read.
|
|
4336
|
+
|
|
4337
|
+
The re-arm surfaces read the journal TWICE to diff what a re-arm appended, and
|
|
4338
|
+
before that echo existed they read it not at all — so `Journal.entries()`' strict
|
|
4339
|
+
UTF-8 decode would turn a corrupt journal into a re-arm the operator can no longer
|
|
4340
|
+
perform, which is strictly worse than the missing echo and a regression against the
|
|
4341
|
+
gesture's own history. Shared by `cli.cmd_resolve` and `TuiApp._do_rearm` rather
|
|
4342
|
+
than living on one of them: the CLI's copy was left unguarded when the TUI's was
|
|
4343
|
+
hardened, and the CLI's echo now runs from a `finally`, where a raise would replace
|
|
4344
|
+
the `RearmError` the operator actually needs to see.
|
|
4345
|
+
|
|
4346
|
+
``None`` rather than ``[]`` because the two callers DIFF two reads. Degrading a
|
|
4347
|
+
failed FIRST read to ``[]`` sets the watermark to zero, and a second read that
|
|
4348
|
+
succeeds then replays every historical `rearm-*`/`stale-restore-*` entry as if this
|
|
4349
|
+
re-arm had just produced it. A caller that cannot establish both ends of the diff
|
|
4350
|
+
must skip the echo, not guess at it.
|
|
4351
|
+
"""
|
|
4352
|
+
try:
|
|
4353
|
+
# Non-mapping lines are dropped HERE so the annotation is true for every
|
|
4354
|
+
# caller: `Journal.entries()` appends `json.loads(line)` with no shape filter,
|
|
4355
|
+
# so a bare `3` or `null` on its own line survives as a non-dict entry and its
|
|
4356
|
+
# `list[dict[str, Any]]` return type is a claim about first-party producers,
|
|
4357
|
+
# not a guarantee — pyright sees `Any` and is satisfied. Both reads apply the
|
|
4358
|
+
# same filter, so the `len(before)` watermark stays exact.
|
|
4359
|
+
return [e for e in Journal(run_dir).entries() if isinstance(e, dict)]
|
|
4360
|
+
except (OSError, UnicodeDecodeError):
|
|
4361
|
+
return None
|
|
4362
|
+
|
|
4363
|
+
|
|
4364
|
+
def _journal_sequence(value: Any) -> tuple[Any, ...]:
|
|
4365
|
+
"""A journal list field read back as a sequence, whatever the line actually held.
|
|
4366
|
+
|
|
4367
|
+
Every read in `rearm_event_notice` runs inside both operator surfaces' `finally`,
|
|
4368
|
+
where a `TypeError` replaces the outcome the operator needs — on the TUI, whose
|
|
4369
|
+
`_do_rearm` runs on Textual's message loop with no `_handle_exception` override,
|
|
4370
|
+
it ends the app. `", ".join` and `len` are the two reads that raise on a shape the
|
|
4371
|
+
journal admits (`"files": 3`, `"files": null`, `[1, 2]`); every sibling read is
|
|
4372
|
+
already `str()`-wrapped or f-string-interpolated and cannot.
|
|
4373
|
+
|
|
4374
|
+
A bare string is deliberately NOT iterated: `", ".join("abc")` renders `"a, b, c"`,
|
|
4375
|
+
which is worse than useless. It is wrapped as a single element instead, and `None`
|
|
4376
|
+
— which `.get(key, default)` returns whenever the key EXISTS holding null, so the
|
|
4377
|
+
default never applies — reads as empty.
|
|
4378
|
+
"""
|
|
4379
|
+
if isinstance(value, (list, tuple)):
|
|
4380
|
+
return tuple(value)
|
|
4381
|
+
return () if value is None else (value,)
|
|
4382
|
+
|
|
4383
|
+
|
|
4384
|
+
def rearm_event_notice(
|
|
4385
|
+
entry: dict[str, Any],
|
|
4386
|
+
) -> tuple[Literal["note", "warning"], str, str] | None:
|
|
4387
|
+
"""`(severity, message, next_step)` for a re-arm record an operator must see.
|
|
4388
|
+
|
|
4389
|
+
ONE table, two surfaces. `cli._echo_rearm_events` prints `message` followed by
|
|
4390
|
+
`next_step`; `TuiApp._do_rearm` shows `message` alone. That split is the whole
|
|
4391
|
+
reason this returns three fields instead of a formatted line: the TUI re-arms and
|
|
4392
|
+
RESUMES in a single gesture, so an instruction to check something "before
|
|
4393
|
+
resuming" is already unactionable by the time it renders — but the finding it
|
|
4394
|
+
reports is not, and dropping the record to avoid the dead imperative is what left
|
|
4395
|
+
the TUI silent on three kinds `resolve` echoed.
|
|
4396
|
+
|
|
4397
|
+
Returns None for journal kinds no operator has to act on, so a caller can walk
|
|
4398
|
+
every new entry and let the table decide.
|
|
4399
|
+
|
|
4400
|
+
Severity is `"note"` or `"warning"`; each surface maps those onto its own channel.
|
|
4401
|
+
"""
|
|
4402
|
+
if not isinstance(entry, dict):
|
|
4403
|
+
return None
|
|
4404
|
+
kind = entry.get("kind", "")
|
|
4405
|
+
if kind == "stale-restore-excluded":
|
|
4406
|
+
files = ", ".join(str(f) for f in _journal_sequence(entry.get("files")))
|
|
4407
|
+
return (
|
|
4408
|
+
"note",
|
|
4409
|
+
f"excluded the abandoned restore's new files from the re-drive baseline: {files}",
|
|
4410
|
+
"",
|
|
4411
|
+
)
|
|
4412
|
+
if kind == "stale-restore-unparseable":
|
|
4413
|
+
return (
|
|
4414
|
+
"warning",
|
|
4415
|
+
f"could not read the abandoned restore patch ({entry.get('patch', '?')}) "
|
|
4416
|
+
"— its new files may be swept into the next commit",
|
|
4417
|
+
"check `git status` before resuming",
|
|
4418
|
+
)
|
|
4419
|
+
if kind == "stale-restore-commits":
|
|
4420
|
+
n = len(_journal_sequence(entry.get("commits")))
|
|
4421
|
+
return (
|
|
4422
|
+
"warning",
|
|
4423
|
+
f"{n} commit(s) sit below the re-drive's new baseline "
|
|
4424
|
+
f"({str(entry.get('old_baseline', '?'))[:12]}..) — if any came from the "
|
|
4425
|
+
"abandoned attempt rather than your resolve, revert them now",
|
|
4426
|
+
"",
|
|
4427
|
+
)
|
|
4428
|
+
if kind == "rearm-baseline-advance-failed":
|
|
4429
|
+
return (
|
|
4430
|
+
"warning",
|
|
4431
|
+
f"could not advance the re-drive baseline ({entry.get('error', '?')}) — it "
|
|
4432
|
+
f"still names {str(entry.get('baseline', '') or '(none)')[:12]}, so the "
|
|
4433
|
+
"re-drive rebuilds against the tree as it stood before your resolve; the "
|
|
4434
|
+
"spec was deliberately NOT re-stamped",
|
|
4435
|
+
"Check the baseline before resuming",
|
|
4436
|
+
)
|
|
4437
|
+
if kind == "rearm-spec-write-unreachable":
|
|
4438
|
+
# ONE kind, TWO remedies, told apart by the `redrive` field its producer writes
|
|
4439
|
+
# — the live isolation mode of the re-drive, which this reader runs too late and
|
|
4440
|
+
# in the wrong process to determine for itself. A record predating the field is
|
|
4441
|
+
# an ISOLATED one: that was the only shape the producer could journal before the
|
|
4442
|
+
# in-place arm existed, so the absent field is a known value, not an unknown.
|
|
4443
|
+
spec = entry.get("spec_file", "?")
|
|
4444
|
+
if str(entry.get("redrive", "isolated") or "isolated") == "in-place":
|
|
4445
|
+
# The mirror shape: `isolation` was edited to `"none"` while the escalation
|
|
4446
|
+
# was paused, so the writes went into the mount the escalated attempt
|
|
4447
|
+
# recorded and the re-drive reads the main checkout instead. Committing is
|
|
4448
|
+
# not the remedy here and naming a branch would be actively wrong — the
|
|
4449
|
+
# in-place re-drive reads a WORKING TREE, so the edit simply has to be made
|
|
4450
|
+
# in the checkout the run resumes into.
|
|
4451
|
+
return (
|
|
4452
|
+
"warning",
|
|
4453
|
+
f"this run's isolation policy changed to `none` while the story was "
|
|
4454
|
+
f"escalated, so the re-arm's spec writes ({spec}) landed in the "
|
|
4455
|
+
"escalated attempt's worktree while the re-drive now runs in the main "
|
|
4456
|
+
"checkout — re-apply the correction to the main checkout's copy of the "
|
|
4457
|
+
"spec or the story re-wedges on the escalated attempt's status",
|
|
4458
|
+
"Correct the spec in the main checkout before resuming",
|
|
4459
|
+
)
|
|
4460
|
+
# The branch is the half an operator cannot infer: the re-drive cuts its fresh
|
|
4461
|
+
# worktree from the run's PINNED target branch, so a correction committed on
|
|
4462
|
+
# whatever the main checkout happens to have checked out is not the one it
|
|
4463
|
+
# reads. Named only when the record carries it — a run predating the field
|
|
4464
|
+
# leaves it empty, and a remedy that names no ref beats one that names a guess.
|
|
4465
|
+
base = str(entry.get("target_branch", "") or "")
|
|
4466
|
+
where = f" on `{base}`" if base else ""
|
|
4467
|
+
return (
|
|
4468
|
+
"warning",
|
|
4469
|
+
f"the re-drive of this story will mount a fresh worktree, so the re-arm's "
|
|
4470
|
+
f"spec writes ({spec}) land in a tree it discards — the re-driven session "
|
|
4471
|
+
"reads the COMMITTED spec, so commit the corrected "
|
|
4472
|
+
f"spec{where} or the story re-wedges on the escalated attempt's status",
|
|
4473
|
+
f"Commit the corrected spec{where} before resuming",
|
|
4474
|
+
)
|
|
4475
|
+
if kind == "rearm-upstream-write-unreachable":
|
|
4476
|
+
# The sentinel counterpart, and ONE remedy rather than the two above: the
|
|
4477
|
+
# producer only reaches this record on the mounting leg, because an in-place
|
|
4478
|
+
# re-drive reads the very checkout `resolve.run_session` ran the agent in. So
|
|
4479
|
+
# there is no `redrive` discriminator to read and no in-place arm to get wrong.
|
|
4480
|
+
#
|
|
4481
|
+
# It names the FOLDER, not a file, because the correction is not one file: the
|
|
4482
|
+
# skill sends the agent to `SPEC.md` or to this story's entry in `stories.yaml`,
|
|
4483
|
+
# and which of the two moved is the agent's choice, not something a journal
|
|
4484
|
+
# reader can recover. Naming both and the folder they sit in is what makes the
|
|
4485
|
+
# remedy actionable without claiming more than the record proves.
|
|
4486
|
+
root = str(entry.get("stories_root", "?"))
|
|
4487
|
+
base = str(entry.get("target_branch", "") or "")
|
|
4488
|
+
where = f" on `{base}`" if base else ""
|
|
4489
|
+
return (
|
|
4490
|
+
"warning",
|
|
4491
|
+
f"the sentinel was cleared, but the re-drive of this story will mount a "
|
|
4492
|
+
f"fresh worktree and re-plan from the COMMITTED tree — the upstream "
|
|
4493
|
+
f"correction in {root} (`SPEC.md` / `stories.yaml`) is uncommitted there, "
|
|
4494
|
+
f"so the re-plan reads the same intent that wedged and mints the sentinel "
|
|
4495
|
+
"again",
|
|
4496
|
+
f"Commit the corrected SPEC.md / stories.yaml{where} before resuming",
|
|
4497
|
+
)
|
|
4498
|
+
if kind == "rearm-spec-flip-skipped":
|
|
4499
|
+
# ONE kind, TWO outcomes, told apart by the flag the producer writes rather
|
|
4500
|
+
# than by anything readable from here: `rearm_escalation` raises `RearmError`
|
|
4501
|
+
# right after journalling this only when the flip failed on the very copy the
|
|
4502
|
+
# re-drive reads. It also journals it — and completes — when that copy is
|
|
4503
|
+
# unreadable from this process, or is a worktree-local file the re-drive
|
|
4504
|
+
# discards. This row used to claim the abort unconditionally, which told an
|
|
4505
|
+
# operator whose re-arm had SUCCEEDED that it "was REFUSED" and sent them to
|
|
4506
|
+
# add a `status:` to a file the re-drive never opens.
|
|
4507
|
+
spec = entry.get("spec_file", "?")
|
|
4508
|
+
status = entry.get("status", "?")
|
|
4509
|
+
if entry.get("refused"):
|
|
4510
|
+
# The message names the refusal rather than predicting a re-wedge, because
|
|
4511
|
+
# there is no re-drive left to wedge — and the next_step is the repair, not
|
|
4512
|
+
# an inspection, for the same reason.
|
|
4513
|
+
return (
|
|
4514
|
+
"warning",
|
|
4515
|
+
f"the recorded spec for this story ({spec}) could not be re-opened to "
|
|
4516
|
+
f"`{status}` — it carries no frontmatter `status:` to set, so the "
|
|
4517
|
+
"re-arm was REFUSED rather than re-driving a session that would wedge "
|
|
4518
|
+
"on the status it reads",
|
|
4519
|
+
"Add a top-level `status:` to the spec, then re-run resolve",
|
|
4520
|
+
)
|
|
4521
|
+
# No next_step, and deliberately: on this leg there is nothing to do to THIS
|
|
4522
|
+
# file. Whether anything is left to do at all is decided by the committed spec,
|
|
4523
|
+
# and `rearm-spec-write-unreachable` — journalled from the same block, on
|
|
4524
|
+
# exactly the legs where the committed spec is not already at the target —
|
|
4525
|
+
# carries that imperative, and holds the resume behind it.
|
|
4526
|
+
return (
|
|
4527
|
+
"warning",
|
|
4528
|
+
f"the recorded spec for this story ({spec}) could not be re-opened to "
|
|
4529
|
+
f"`{status}` — the re-arm was NOT refused, because that copy is not what "
|
|
4530
|
+
"the re-driven session reads: it mounts a fresh worktree and reads the "
|
|
4531
|
+
"COMMITTED spec",
|
|
4532
|
+
"",
|
|
4533
|
+
)
|
|
4534
|
+
if kind == "rearm-baseline-restamp-skipped":
|
|
4535
|
+
return (
|
|
4536
|
+
"warning",
|
|
4537
|
+
f"the recorded spec for this story ({entry.get('spec_file', '?')}) is not a "
|
|
4538
|
+
"readable file from here, so the baseline re-stamp was skipped — the spec "
|
|
4539
|
+
"still names the escalated attempt's baseline",
|
|
4540
|
+
"Check the recorded spec path before resuming",
|
|
4541
|
+
)
|
|
4542
|
+
if kind == "rearm-baseline-restamped":
|
|
4543
|
+
head = (
|
|
4544
|
+
f"re-stamped the spec baseline "
|
|
4545
|
+
f"{str(entry.get('overwritten', '?'))[:12]}.. -> "
|
|
4546
|
+
f"{str(entry.get('baseline', '?'))[:12]}.."
|
|
4547
|
+
)
|
|
4548
|
+
# NOT differentiated on the `restore` flag any more. That split predated the
|
|
4549
|
+
# record's condition moving to `overwritten != old_baseline` (compared against
|
|
4550
|
+
# what the RUN recorded, not against the just-advanced value): the record now
|
|
4551
|
+
# fires ONLY when the spec claimed a baseline the run never recorded, which is
|
|
4552
|
+
# equally exceptional on both legs. Keeping the split meant the patch-restore
|
|
4553
|
+
# leg's real divergence was the one downgraded to a note. The flag stays ON the
|
|
4554
|
+
# record because it says which leg produced it — not how routine it is.
|
|
4555
|
+
return (
|
|
4556
|
+
"warning",
|
|
4557
|
+
f"{head} — the spec claimed a DIFFERENT baseline than the run recorded, "
|
|
4558
|
+
"and this re-stamp is the only trace of it; the gate can no longer report "
|
|
4559
|
+
"that divergence",
|
|
4560
|
+
"",
|
|
4561
|
+
)
|
|
4562
|
+
return None
|
|
4563
|
+
|
|
4564
|
+
|
|
4565
|
+
def rearm_holds_the_resume(entry: dict[str, Any]) -> bool:
|
|
4566
|
+
"""True for a re-arm record whose remedy has to land BEFORE the re-drive reads the
|
|
4567
|
+
tree — so a surface that re-arms and resumes in ONE gesture must stop after the
|
|
4568
|
+
re-arm and leave `froid-loop resume` to the operator.
|
|
4569
|
+
|
|
4570
|
+
TWO kinds qualify, and the discriminator is PROOF, not urgency.
|
|
4571
|
+
`rearm-spec-write-unreachable` is written only once `_redrive_spec_status` has
|
|
4572
|
+
established that the committed spec does NOT carry the status the re-drive routes
|
|
4573
|
+
on, and only for a spec the working-tree flip cannot reach. Resuming on it is not
|
|
4574
|
+
risky, it is futile: the re-drive discards the worktree, mounts a fresh one from
|
|
4575
|
+
git, and step-01 reads a status it cannot route — `unrecognized status in existing
|
|
4576
|
+
story file` halts it blocked, and the escalation is spent. The record's own
|
|
4577
|
+
next_step already said "commit the corrected spec before resuming"; both default
|
|
4578
|
+
surfaces then resumed in the same breath, which made the imperative unactionable at
|
|
4579
|
+
the moment it rendered. The interactive resolve agent cannot close that gap either
|
|
4580
|
+
— its skill forbids it from committing.
|
|
4581
|
+
|
|
4582
|
+
`rearm-upstream-write-unreachable` earns it the same way on the sentinel path,
|
|
4583
|
+
where there is no spec write to measure at all: the sentinel is cleared by
|
|
4584
|
+
deletion, and the correction that stops it recurring sits upstream in `SPEC.md` /
|
|
4585
|
+
`stories.yaml`. Its proof is `_redrive_reads_the_upstream_artifacts`, which fires
|
|
4586
|
+
the record only while the ref the re-drive mounts from does NOT already hold this
|
|
4587
|
+
checkout's copy of those two files — so, exactly as above, resuming is not risky
|
|
4588
|
+
but futile: the re-drive re-plans from a tree that never saw the correction and
|
|
4589
|
+
mints the same sentinel again.
|
|
4590
|
+
|
|
4591
|
+
The other warnings stay advisory and do NOT hold. `stale-restore-commits`,
|
|
4592
|
+
`stale-restore-unparseable` and `rearm-baseline-advance-failed` each report
|
|
4593
|
+
something an operator may need to act on, but none of them PROVES the re-drive
|
|
4594
|
+
cannot route, and holding on a maybe would turn the ordinary degrade path into a
|
|
4595
|
+
two-command gesture for an outcome nothing decided.
|
|
4596
|
+
|
|
4597
|
+
Not folded into `rearm_event_notice`'s tuple, because they are different questions
|
|
4598
|
+
asked of the same entry: that table answers "what do I tell the operator", this
|
|
4599
|
+
answers "may this gesture still resume". Both surfaces ask both, in one walk.
|
|
4600
|
+
"""
|
|
4601
|
+
return isinstance(entry, dict) and entry.get("kind") in (
|
|
4602
|
+
"rearm-spec-write-unreachable",
|
|
4603
|
+
"rearm-upstream-write-unreachable",
|
|
4604
|
+
)
|
|
4605
|
+
|
|
4606
|
+
|
|
4607
|
+
def _stale_restore_residue(
|
|
4608
|
+
repo: Path,
|
|
4609
|
+
journal: Journal,
|
|
4610
|
+
story_key: str,
|
|
4611
|
+
old_latch: str | None,
|
|
4612
|
+
old_baseline: str | None,
|
|
4613
|
+
) -> set[str]:
|
|
4614
|
+
"""The untracked files an abandoned patch-restore attempt left in the tree —
|
|
4615
|
+
to be subtracted from the re-arm's refreshed `baseline_untracked` (issue #90).
|
|
4616
|
+
|
|
4617
|
+
Empty when no restore was latched. Deliberately *not* a `git apply -R`: the
|
|
4618
|
+
re-drive's own reset already reverts the patch's tracked hunks, an `apply -R`
|
|
4619
|
+
fails outright on any drift the resolve session introduced, and it misbehaves
|
|
4620
|
+
on the committed variant below. Only the patch's new files are durable
|
|
4621
|
+
contamination, and naming them is enough — `verify.safe_rollback` deletes
|
|
4622
|
+
whatever the refreshed snapshot stops blessing.
|
|
4623
|
+
|
|
4624
|
+
Also journals (warn-only) the commits sitting between the OLD baseline and the
|
|
4625
|
+
new one: a commit the escalated re-drive session made now becomes the next
|
|
4626
|
+
re-drive's permanent starting point, and no reset revisits it. It is not
|
|
4627
|
+
mechanically reversible — the resolve session's own blessed commits live in the
|
|
4628
|
+
same range and reverting those would claw back the human's resolution — so the
|
|
4629
|
+
human is the classifier. `froid-loop resolve` echoes these to stderr.
|
|
4630
|
+
|
|
4631
|
+
Best-effort throughout: a deleted or unreadable patch, a non-repo project, a
|
|
4632
|
+
bad old baseline — none may wedge a resolve. Every failure degrades to the
|
|
4633
|
+
pre-#90 behavior and says so in the journal.
|
|
4634
|
+
"""
|
|
4635
|
+
if not old_latch:
|
|
4636
|
+
return set()
|
|
4637
|
+
patch_path = verify.resolve_restore_path(old_latch, repo)
|
|
4638
|
+
|
|
4639
|
+
residue: set[str] = set()
|
|
4640
|
+
try:
|
|
4641
|
+
residue = verify.patch_new_files(patch_path)
|
|
4642
|
+
except (OSError, UnicodeDecodeError) as e:
|
|
4643
|
+
# degrade to the pre-#90 snapshot rather than wedge the resolve
|
|
4644
|
+
journal.append(
|
|
4645
|
+
"stale-restore-unparseable",
|
|
4646
|
+
story_key=story_key,
|
|
4647
|
+
patch=str(patch_path),
|
|
4648
|
+
error=f"{e.__class__.__name__}: {e}",
|
|
4649
|
+
)
|
|
4650
|
+
else:
|
|
4651
|
+
if residue:
|
|
4652
|
+
journal.append(
|
|
4653
|
+
"stale-restore-excluded",
|
|
4654
|
+
story_key=story_key,
|
|
4655
|
+
patch=str(patch_path),
|
|
4656
|
+
files=sorted(residue),
|
|
4657
|
+
)
|
|
4658
|
+
|
|
4659
|
+
# Independent of the parse above — an unreadable patch must not also cost the
|
|
4660
|
+
# human the only notice they get about the committed variant.
|
|
4661
|
+
if old_baseline:
|
|
4662
|
+
try:
|
|
4663
|
+
shas = verify.commits_above(repo, old_baseline)
|
|
4664
|
+
except Exception: # nosec B110 - warn-only, must not fail re-arm
|
|
4665
|
+
shas = []
|
|
4666
|
+
if shas:
|
|
4667
|
+
journal.append(
|
|
4668
|
+
"stale-restore-commits",
|
|
4669
|
+
story_key=story_key,
|
|
4670
|
+
old_baseline=old_baseline,
|
|
4671
|
+
commits=shas,
|
|
4672
|
+
)
|
|
4673
|
+
return residue
|
|
4674
|
+
|
|
4675
|
+
|
|
4676
|
+
def _sentinel_condition(spec_path: Path, story_key: str) -> str | None:
|
|
4677
|
+
"""The blocking condition (``unresolved`` / ``ambiguous``) iff ``spec_path`` is
|
|
4678
|
+
a fixed-slug pre-planning-halt sentinel for ``story_key``, else None."""
|
|
4679
|
+
from .stories import SENTINEL_SLUGS
|
|
4680
|
+
|
|
4681
|
+
for slug in SENTINEL_SLUGS:
|
|
4682
|
+
if spec_path.name == f"{story_key}-{slug}.md":
|
|
4683
|
+
return slug
|
|
4684
|
+
return None
|
|
4685
|
+
|
|
4686
|
+
|
|
4687
|
+
def _clear_sentinel(
|
|
4688
|
+
run_dir: Path, journal: Journal, spec_path: Path, story_key: str, sentinel_kind: str
|
|
4689
|
+
) -> None:
|
|
4690
|
+
"""Preserve a copy of the sentinel under ``{run_dir}/sentinels/`` (a write-only
|
|
4691
|
+
breadcrumb of what blocked planning), journal ``sentinel-cleared`` — carrying
|
|
4692
|
+
both the fixed slug (``sentinel_kind``) and the *recorded blocking condition*
|
|
4693
|
+
parsed from the sentinel's ``## Auto Run Result`` (the reason planning halted) —
|
|
4694
|
+
then delete the sentinel so the next dispatch is clean."""
|
|
4695
|
+
from .stories import recorded_blocking_condition
|
|
4696
|
+
|
|
4697
|
+
dest_dir = run_dir / "sentinels"
|
|
4698
|
+
dest_dir.mkdir(parents=True, exist_ok=True)
|
|
4699
|
+
condition = ""
|
|
4700
|
+
if spec_path.is_file():
|
|
4701
|
+
try:
|
|
4702
|
+
condition = recorded_blocking_condition(spec_path.read_text(encoding="utf-8"))
|
|
4703
|
+
except (OSError, UnicodeDecodeError):
|
|
4704
|
+
# An unreadable/binary sentinel still gets preserved+deleted so re-arm
|
|
4705
|
+
# completes; we just journal an empty blocking condition.
|
|
4706
|
+
condition = ""
|
|
4707
|
+
shutil.copy2(spec_path, dest_dir / spec_path.name)
|
|
4708
|
+
spec_path.unlink()
|
|
4709
|
+
journal.append(
|
|
4710
|
+
"sentinel-cleared",
|
|
4711
|
+
story_key=story_key,
|
|
4712
|
+
sentinel_kind=sentinel_kind,
|
|
4713
|
+
condition=condition,
|
|
4714
|
+
sentinel=spec_path.name,
|
|
4715
|
+
)
|