froid-loop 0.11.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. froid_loop/__init__.py +11 -0
  2. froid_loop/__main__.py +12 -0
  3. froid_loop/adapters/__init__.py +3 -0
  4. froid_loop/adapters/base.py +254 -0
  5. froid_loop/adapters/entrypoints.py +63 -0
  6. froid_loop/adapters/env_fault.py +290 -0
  7. froid_loop/adapters/generic.py +2013 -0
  8. froid_loop/adapters/mock.py +49 -0
  9. froid_loop/adapters/multiplexer.py +914 -0
  10. froid_loop/adapters/opencode_http.py +1687 -0
  11. froid_loop/adapters/profile.py +650 -0
  12. froid_loop/adapters/psmux_backend.py +1428 -0
  13. froid_loop/adapters/registry.py +322 -0
  14. froid_loop/adapters/tmux_backend.py +35 -0
  15. froid_loop/adapters/tmux_base.py +630 -0
  16. froid_loop/checks.py +187 -0
  17. froid_loop/cli.py +5041 -0
  18. froid_loop/data/__init__.py +0 -0
  19. froid_loop/data/froid_loop_hook.py +228 -0
  20. froid_loop/data/froid_loop_probe_hook.py +88 -0
  21. froid_loop/data/plugins/example/plugin.toml +21 -0
  22. froid_loop/data/plugins/tea/plugin.toml +184 -0
  23. froid_loop/data/plugins/tea/tea_plugin.py +258 -0
  24. froid_loop/data/plugins/unity/plugin.toml +140 -0
  25. froid_loop/data/plugins/unity/unity_assets/FroidLoop.Unity.Editor.asmdef +16 -0
  26. froid_loop/data/plugins/unity/unity_assets/FroidLoop.Unity.Editor.asmdef.meta +7 -0
  27. froid_loop/data/plugins/unity/unity_assets/SceneAutoSaveGuard.cs +221 -0
  28. froid_loop/data/plugins/unity/unity_assets/SceneAutoSaveGuard.cs.meta +11 -0
  29. froid_loop/data/plugins/unity/unity_assets/_folders/Editor.meta +8 -0
  30. froid_loop/data/plugins/unity/unity_assets/_folders/FroidLoop.meta +8 -0
  31. froid_loop/data/plugins/unity/unity_cleanup.py +125 -0
  32. froid_loop/data/plugins/unity/unity_dialog_probe.py +239 -0
  33. froid_loop/data/plugins/unity/unity_facts.md +17 -0
  34. froid_loop/data/plugins/unity/unity_plugin.py +415 -0
  35. froid_loop/data/plugins/unity/unity_quiesce.py +234 -0
  36. froid_loop/data/plugins/unity/unity_ready.py +230 -0
  37. froid_loop/data/plugins/unity/unity_seed_assets.py +298 -0
  38. froid_loop/data/plugins/unity/unity_setup.py +551 -0
  39. froid_loop/data/plugins/unity/unity_teardown.py +362 -0
  40. froid_loop/data/profiles/antigravity.toml +52 -0
  41. froid_loop/data/profiles/claude.toml +85 -0
  42. froid_loop/data/profiles/codex.toml +22 -0
  43. froid_loop/data/profiles/copilot.toml +52 -0
  44. froid_loop/data/profiles/gemini.toml +26 -0
  45. froid_loop/data/profiles/opencode.toml +54 -0
  46. froid_loop/data/settings/core.toml +458 -0
  47. froid_loop/data/skills/README.md +93 -0
  48. froid_loop/data/skills/froid-loop-resolve/SKILL.md +288 -0
  49. froid_loop/data/skills/froid-loop-setup/SKILL.md +161 -0
  50. froid_loop/data/skills/froid-loop-setup/assets/module-help.csv +3 -0
  51. froid_loop/data/skills/froid-loop-setup/assets/module.yaml +19 -0
  52. froid_loop/data/skills/froid-loop-sweep/SKILL.md +100 -0
  53. froid_loop/data/skills/froid-loop-sweep/automation-mode.md +127 -0
  54. froid_loop/data/skills/froid-loop-sweep/deferred-work-format.md +302 -0
  55. froid_loop/data/skills/froid-loop-sweep/migration-mode.md +86 -0
  56. froid_loop/decisions.py +202 -0
  57. froid_loop/deferredwork.py +2282 -0
  58. froid_loop/devcontract.py +892 -0
  59. froid_loop/diagnostics.py +1104 -0
  60. froid_loop/documents.py +532 -0
  61. froid_loop/engine.py +7732 -0
  62. froid_loop/envvars.py +111 -0
  63. froid_loop/escalation.py +225 -0
  64. froid_loop/events.py +266 -0
  65. froid_loop/fences.py +103 -0
  66. froid_loop/froidconfig.py +226 -0
  67. froid_loop/frontmatter.py +526 -0
  68. froid_loop/gates.py +133 -0
  69. froid_loop/install.py +2936 -0
  70. froid_loop/journal.py +178 -0
  71. froid_loop/machine.py +148 -0
  72. froid_loop/model.py +898 -0
  73. froid_loop/operatoractions.py +474 -0
  74. froid_loop/platform_util.py +1490 -0
  75. froid_loop/plugins/__init__.py +64 -0
  76. froid_loop/plugins/bus.py +259 -0
  77. froid_loop/plugins/context.py +319 -0
  78. froid_loop/plugins/loader.py +145 -0
  79. froid_loop/plugins/manifest.py +279 -0
  80. froid_loop/plugins/model.py +296 -0
  81. froid_loop/plugins/registry.py +245 -0
  82. froid_loop/plugins/trust.py +75 -0
  83. froid_loop/policy.py +1569 -0
  84. froid_loop/probe.py +1044 -0
  85. froid_loop/process_host.py +408 -0
  86. froid_loop/recovery_flow.py +1561 -0
  87. froid_loop/resolve.py +283 -0
  88. froid_loop/runs.py +4715 -0
  89. froid_loop/runsetup.py +1293 -0
  90. froid_loop/sanitize.py +593 -0
  91. froid_loop/settings_schema.py +276 -0
  92. froid_loop/signals.py +160 -0
  93. froid_loop/sprintstatus.py +609 -0
  94. froid_loop/statemachine.py +57 -0
  95. froid_loop/stories.py +615 -0
  96. froid_loop/stories_engine.py +796 -0
  97. froid_loop/sweep.py +1892 -0
  98. froid_loop/tokens.py +196 -0
  99. froid_loop/tui/__init__.py +11 -0
  100. froid_loop/tui/app.py +1584 -0
  101. froid_loop/tui/data.py +840 -0
  102. froid_loop/tui/launch.py +1003 -0
  103. froid_loop/tui/screens/__init__.py +1 -0
  104. froid_loop/tui/screens/dashboard.py +1071 -0
  105. froid_loop/tui/screens/modals.py +943 -0
  106. froid_loop/tui/screens/settings_screen.py +477 -0
  107. froid_loop/tui/settings.py +135 -0
  108. froid_loop/tui/widgets.py +981 -0
  109. froid_loop/verify.py +4545 -0
  110. froid_loop/workspace.py +320 -0
  111. froid_loop/worktree_flow.py +2301 -0
  112. froid_loop-0.11.1.dist-info/METADATA +728 -0
  113. froid_loop-0.11.1.dist-info/RECORD +116 -0
  114. froid_loop-0.11.1.dist-info/WHEEL +4 -0
  115. froid_loop-0.11.1.dist-info/entry_points.txt +2 -0
  116. froid_loop-0.11.1.dist-info/licenses/LICENSE +30 -0
froid_loop/model.py ADDED
@@ -0,0 +1,898 @@
1
+ """Core data model: story lifecycle phases, per-task records, run state."""
2
+
3
+ # Strict-checked under #245 Stage 2, with the two rules below relaxed for this
4
+ # file only. `reportUnknownVariableType`: the dataclass collection fields use the
5
+ # idiomatic `field(default_factory=list|dict)`, which pyright can only infer as
6
+ # `list[Unknown]` / `dict[Unknown, Unknown]` (it does not fold the declared
7
+ # annotation back into the factory) though the fields are correctly typed.
8
+ # `reportUnknownArgumentType`: `from_dict` / snapshot readers pull values out of
9
+ # run-persisted `dict[str, Any]`, so isinstance-narrowing an `Any` value yields
10
+ # Unknown at that boundary. Both are inherent to the persistence edge, not
11
+ # annotation drift; every other strict rule stays on.
12
+ # pyright: reportUnknownArgumentType=false, reportUnknownVariableType=false
13
+
14
+ from __future__ import annotations
15
+
16
+ import base64
17
+ from copy import deepcopy
18
+ from dataclasses import dataclass, field
19
+ from enum import StrEnum
20
+ from pathlib import Path
21
+ from typing import Any
22
+
23
+
24
+ class Phase(StrEnum):
25
+ PENDING = "pending"
26
+ DEV_RUNNING = "dev-running"
27
+ DEV_VERIFY = "dev-verify"
28
+ REVIEW_RUNNING = "review-running"
29
+ REVIEW_VERIFY = "review-verify"
30
+ COMMITTING = "committing"
31
+ # sweep-only: the triage session classifying open deferred-work entries
32
+ TRIAGE_RUNNING = "triage-running"
33
+ TRIAGE_VERIFY = "triage-verify"
34
+ DONE = "done"
35
+ DEFERRED = "deferred"
36
+ ESCALATED = "escalated"
37
+ # the story's agent-doable work is finished and COMMITTED, but its acceptance
38
+ # criteria include external actions only a human can perform (buy a domain,
39
+ # publish a DNS record). Terminal like DONE — the run moves on rather than
40
+ # halting — but not DONE: the outstanding actions are recorded on the task.
41
+ # Deliberately NOT a pause: an operator-owed story must never block the
42
+ # stories behind it (contrast the spec's `Block If:` -> blocked -> CRITICAL
43
+ # -> pause channel, which is run-halting by design).
44
+ AWAITING_OPERATOR = "awaiting-operator"
45
+
46
+
47
+ # Terminal = the orchestrator will not drive this story further in this run. It
48
+ # says nothing about success: DONE and AWAITING_OPERATOR carry a commit, DEFERRED
49
+ # and ESCALATED do not.
50
+ TERMINAL_PHASES = frozenset({Phase.DONE, Phase.DEFERRED, Phase.ESCALATED, Phase.AWAITING_OPERATOR})
51
+
52
+ # Pause stages recorded in RunState.paused_stage
53
+ PAUSE_SPEC_APPROVAL = "spec-approval"
54
+ PAUSE_EPIC_BOUNDARY = "epic-boundary"
55
+ PAUSE_ESCALATION = "escalation"
56
+ # Raised by Engine._refuse_gated_story: the picked story is named by the `gate:`
57
+ # line of a deferred-work entry that has not landed. Produced before the story is
58
+ # recorded in state.tasks, so a resume re-picks it and re-reads the ledger.
59
+ PAUSE_STORY_GATE = "story-gate"
60
+ # stories-mode HITL checkpoints (independent per story). PLAN fires after a
61
+ # spec_checkpoint story's plan-halt leg (ready-for-dev, awaiting human plan
62
+ # review before implementation); STORY fires after a done_checkpoint story's
63
+ # commit (skip-if-last). Both re-arm through the same resume path.
64
+ PAUSE_PLAN_CHECKPOINT = "plan-checkpoint"
65
+ PAUSE_STORY_CHECKPOINT = "story-checkpoint"
66
+
67
+ # Reasons recorded in RunState.sweeps_refused (trigger -> reason). A CLOSED
68
+ # vocabulary of short slugs, deliberately not a formatted exception: `froid-loop
69
+ # diagnose` renders run state through `sanitize.guard`, which *raises*
70
+ # LeakDetected on a home-path hit rather than redacting it (genuine PII never
71
+ # auto-repairs — sanitize.py). A free-form `str(e)` here would therefore make the
72
+ # dump fail outright on exactly the runs worth dumping, and the per-value
73
+ # `looks_like_identifier` filter would blank it anyway. Add a slug, never a
74
+ # message.
75
+ SWEEP_REFUSED_NOT_STARTED = "not-started" # the launch raised before a child existed
76
+ SWEEP_REFUSED_FAILED = "failed" # a child started, then failed
77
+ SWEEP_REFUSED_DIRTY = "dirty" # the worktree was unclean, or `git status` faulted
78
+
79
+
80
+ @dataclass
81
+ class TokenUsage:
82
+ input_tokens: int = 0
83
+ output_tokens: int = 0
84
+ cache_read_tokens: int = 0
85
+ cache_creation_tokens: int = 0
86
+
87
+ def add(self, other: "TokenUsage") -> None:
88
+ self.input_tokens += other.input_tokens
89
+ self.output_tokens += other.output_tokens
90
+ self.cache_read_tokens += other.cache_read_tokens
91
+ self.cache_creation_tokens += other.cache_creation_tokens
92
+
93
+ @property
94
+ def total(self) -> int:
95
+ return (
96
+ self.input_tokens
97
+ + self.output_tokens
98
+ + self.cache_read_tokens
99
+ + self.cache_creation_tokens
100
+ )
101
+
102
+ def weighted_total(self, cache_read_weight: float) -> int:
103
+ """Cost-proportional total: cache reads are billed at ~0.1x base input
104
+ on all supported vendors (Anthropic/OpenAI/Gemini, June 2026), so raw
105
+ totals mostly measure context re-reads; the budget discounts them."""
106
+ return (
107
+ self.input_tokens
108
+ + self.output_tokens
109
+ + self.cache_creation_tokens
110
+ + round(self.cache_read_tokens * cache_read_weight)
111
+ )
112
+
113
+ def to_dict(self) -> dict[str, int]:
114
+ return {
115
+ "input_tokens": self.input_tokens,
116
+ "output_tokens": self.output_tokens,
117
+ "cache_read_tokens": self.cache_read_tokens,
118
+ "cache_creation_tokens": self.cache_creation_tokens,
119
+ }
120
+
121
+ @classmethod
122
+ def from_dict(cls, d: dict[str, Any]) -> "TokenUsage":
123
+ return cls(
124
+ input_tokens=int(d.get("input_tokens", 0)),
125
+ output_tokens=int(d.get("output_tokens", 0)),
126
+ cache_read_tokens=int(d.get("cache_read_tokens", 0)),
127
+ cache_creation_tokens=int(d.get("cache_creation_tokens", 0)),
128
+ )
129
+
130
+
131
+ @dataclass
132
+ class SessionRecord:
133
+ task_id: str
134
+ role: str # "dev" | "review"
135
+ status: str # SessionResult.status
136
+ # resolved adapter identity for the session (#153 phase 1). adapter "" means
137
+ # the record predates identity stamping; adapter set with model "" means the
138
+ # session ran the CLI profile's default model (no explicit model override).
139
+ adapter: str = ""
140
+ model: str = ""
141
+ session_id: str | None = None
142
+ transcript_path: str | None = None
143
+ usage: TokenUsage | None = None
144
+ # the session's parsed result payload, persisted so a durably-saved
145
+ # completed session is actionable on resume, not just forensics
146
+ result_json: dict[str, Any] | None = None
147
+
148
+ def to_dict(self) -> dict[str, Any]:
149
+ return {
150
+ "task_id": self.task_id,
151
+ "role": self.role,
152
+ "status": self.status,
153
+ "adapter": self.adapter,
154
+ "model": self.model,
155
+ "session_id": self.session_id,
156
+ "transcript_path": self.transcript_path,
157
+ "usage": self.usage.to_dict() if self.usage else None,
158
+ "result_json": self.result_json,
159
+ }
160
+
161
+ @classmethod
162
+ def from_dict(cls, d: dict[str, Any]) -> "SessionRecord":
163
+ usage = d.get("usage")
164
+ return cls(
165
+ task_id=d["task_id"],
166
+ role=d["role"],
167
+ status=d["status"],
168
+ adapter=str(d.get("adapter", "")),
169
+ model=str(d.get("model", "")),
170
+ session_id=d.get("session_id"),
171
+ transcript_path=d.get("transcript_path"),
172
+ usage=TokenUsage.from_dict(usage) if usage else None,
173
+ result_json=d.get("result_json"),
174
+ )
175
+
176
+
177
+ def _rebased_on(path: str | None, root: Path) -> str | None:
178
+ """One persisted spec path, re-anchored on `root`; absolute values pass through.
179
+
180
+ Split out so `StoryTask.rebase_spec_paths_on` states the rule once per field
181
+ without repeating the guard, and so the guard itself is unmissable: the
182
+ is-absolute test is what keeps an out-of-mount spec (persisted verbatim by
183
+ `_serialized_worktree_path`) from being joined onto a root that does not
184
+ contain it.
185
+ """
186
+ if not path or Path(path).is_absolute():
187
+ return path
188
+ return str(root / path)
189
+
190
+
191
+ @dataclass
192
+ class StoryTask:
193
+ story_key: str
194
+ epic: int
195
+ phase: Phase = Phase.PENDING
196
+ attempt: int = 0
197
+ review_cycle: int = 0
198
+ # count of review rounds granted *solely* because a completed round finalized
199
+ # the story (status: done) yet still set `followup_review_recommended: true`.
200
+ # Bounded by limits.max_followup_reviews: once spent, the next such round
201
+ # force-converges (verify → refile the recommendation to the ledger → commit)
202
+ # rather than burning another cycle. Reset to 0 by runs.rearm_escalation so a
203
+ # human-resolved re-drive gets a fresh damping budget. Survives the round-trip.
204
+ followup_reviews_spent: int = 0
205
+ # How many times a human re-arm (`runs.rearm_escalation`) has re-opened this
206
+ # task. Re-arm resets `attempt` to 0 and the next dispatch bumps it back to 1,
207
+ # so without a discriminator the re-minted session task_id is byte-equal to a
208
+ # record the ABANDONED attempt already appended to the append-only `sessions`
209
+ # list — and `Engine._resumable_session`, which matches on that id, replays the
210
+ # abandoned attempt's verdict for the fresh one (#705). Feeds
211
+ # `engine._session_task_id`, which emits the suffix only above zero, so every
212
+ # id already on disk stays byte-identical across the upgrade. `task.sessions`
213
+ # is deliberately NOT cleared at re-arm: the run-dir audit trail it indexes is
214
+ # read by a second resolve cycle.
215
+ generation: int = 0
216
+ # set from the froid-build-auto session's `followup_review_recommended`
217
+ # frontmatter (PR #2505): when True and review.trigger = "recommended", the
218
+ # orchestrator runs a follow-up review pass (froid-build-auto re-invoked on the
219
+ # done spec); otherwise it skips it.
220
+ followup_review_recommended: bool = False
221
+ baseline_commit: str | None = None
222
+ # untracked, non-ignored paths present at baseline capture (repo-relative
223
+ # posix). On rollback only paths NOT in this set are removed, so files the
224
+ # user already had on disk are never deleted. None = pre-upgrade run (no
225
+ # snapshot); rollback then removes no untracked files at all.
226
+ baseline_untracked: list[str] | None = None
227
+ # Deferred-work bookkeeping is persisted before its readers land so an older
228
+ # state.json remains resumable throughout the forward-port. The nullable
229
+ # snapshot text and its captured flag are deliberately separate: None means
230
+ # "no ledger existed", while False means "no snapshot was taken".
231
+ baseline_ledger_digest: str | None = None
232
+ pre_harvest_ledger: str | None = None
233
+ pre_harvest_ledger_captured: bool = False
234
+ # Digest of the last ledger state THIS engine left on disk: the snapshot's
235
+ # own text at capture, refreshed to the post-append bytes once the harvest
236
+ # writes. It is the compare-and-set anchor `_restore_ledger` uses to tell
237
+ # its own retractable write from a concurrent writer's (#286), which is why
238
+ # it is persisted rather than kept in memory: a crash replay must be able to
239
+ # recognize the dead attempt's append still sitting on disk.
240
+ post_engine_ledger_digest: str | None = None
241
+ harvest_wrote_ledger: bool = False
242
+ ledger_changed_before_harvest: bool = False
243
+ # JSON-native containers only; callers persist these through state.json.
244
+ harvested_deferrals: list[dict[str, Any]] = field(default_factory=list)
245
+ bundle_closes_intended: list[str] = field(default_factory=list)
246
+ # `append_entry` kwargs for review-budget follow-ups this task filed into the
247
+ # ACTIVE workspace's ledger, which under isolation is the unit worktree's.
248
+ refiled_followups: list[dict[str, Any]] = field(default_factory=list)
249
+ # Deferred-work ids a story DECLARED it closes (`closes_deferred:`), recorded at
250
+ # the commit boundary. Same ledger, same isolation problem: a gitignored path
251
+ # never merges out of the unit worktree, so the flip has to be re-applied.
252
+ story_closes_intended: list[str] = field(default_factory=list)
253
+ # The sprint-status stage `_post_dev_state_sync` REQUESTED for this story, or
254
+ # None when it never ran (sweep bundles, stories mode, the legacy path). Same
255
+ # isolation problem as the ledger payloads above, one file over: under
256
+ # `isolation = "worktree"` that advance lands on the unit worktree's board,
257
+ # which for a gitignored board is a seeded copy shielded from the unit commit,
258
+ # so the post-merge carry has to re-apply it. Latest-wins — a scalar, not a
259
+ # list, because the board holds one stage per story and only the accepted
260
+ # attempt's stage can ever be carried.
261
+ board_advance_intended: str | None = None
262
+ # Index of the append-only primary dev SessionRecord whose initial decision or
263
+ # later verify-repair result durably returned PROCEED. Attempt numbers can be
264
+ # reused after a human re-arm, so the exact record occurrence is the acceptance
265
+ # identity; None is legacy/unarmed.
266
+ accepted_dev_session_index: int | None = None
267
+ harvest_carry_commit_pending: bool = False
268
+ isolated_ledger_carried: bool = False
269
+ spec_file: str | None = None
270
+ # The spec owned by the current/last dispatched dev attempt. Unlike
271
+ # ``spec_file`` (the accepted/result artifact), this is bound before launch
272
+ # so recovery can identify an attempt's lifecycle-only residue after a crash.
273
+ dispatched_spec_file: str | None = None
274
+ # Byte-exact input contents of ``dispatched_spec_file`` for the current retry
275
+ # chain. The JSON representation is base64, so CRLF and non-UTF-8 bytes survive
276
+ # a crash/resume round-trip. The chain's first bound input is retained across
277
+ # fixable repairs, then restored before a fresh-baseline retry; in a resolved
278
+ # re-drive this is the operator-corrected spec, so child-authored body edits can
279
+ # never become the retained correction. Cleared after successful commit. None =
280
+ # unbound attempt, retired chain, or legacy state.
281
+ dispatched_spec_snapshot: bytes | None = None
282
+ commit_sha: str | None = None
283
+ # the external, human-only actions this story still owes when it parks at
284
+ # Phase.AWAITING_OPERATOR — one free-text instruction per entry, as the dev
285
+ # session enumerated them in the spec's `operator_actions:` frontmatter.
286
+ # Plain strings in v1: a per-action deterministic `check:` command is a
287
+ # deliberate v2 question, and strings-now/objects-later is the cheaper
288
+ # migration than the reverse. Empty on every other phase: owing at least one
289
+ # action is what selects AWAITING_OPERATOR over DONE when the park path picks
290
+ # a committing story's final phase. That choice is the only enforcement —
291
+ # nothing validates this field on load or on write, so a task carrying the
292
+ # phase with no actions is unreachable rather than rejected. Survives the
293
+ # resume serialization round-trip (it is the durable record of what the human
294
+ # owes, and nothing re-derives it once the session that wrote the spec is
295
+ # gone).
296
+ operator_actions: list[str] = field(default_factory=list)
297
+ defer_reason: str | None = None
298
+ # the recovery ref this attempt's work was parked on by the last auto-rollback
299
+ # — an `attempt-preserve/*` branch (commits above baseline) or, when the tree
300
+ # was also dirty, the `refs/attempt-preserve-dirty/*` snapshot, which is
301
+ # parented at the attempt's HEAD and therefore subsumes the branch (last
302
+ # writer wins, so one `git merge --ff-only <ref>` recovers the whole attempt
303
+ # — unless `preserve_partial` is set). Set by RecoveryFlow, cleared at the top
304
+ # of every auto-rollback so it can never name a *previous* attempt's ref; read
305
+ # by `_defer` (notification) and projected into `status`. None = the last
306
+ # auto-rollback parked nothing (no commits above baseline and a clean or
307
+ # uncapturable tree, or the ref failed to take). Isolation-INDEPENDENT: a unit
308
+ # worktree's own dev-retry rollback parks on the same shared refs, so a
309
+ # deferred isolated unit can carry BOTH a kept-failed branch (the final
310
+ # attempt) and a preserve_ref (an earlier, rolled-back one) — `_defer` names
311
+ # both. The unit branch itself is never written here: a live branch is not a
312
+ # parked snapshot. Not cleared on success — a mid-retry rollback's breadcrumb
313
+ # stays readable. Survives the resume serialization round-trip.
314
+ preserve_ref: str | None = None
315
+ # set when the auto-rollback's *worktree* snapshot was attempted and raised
316
+ # (journalled `attempt-worktree-preserve-failed`), so `preserve_ref` names an
317
+ # `attempt-preserve/*` commits branch ALONE and the reset that followed
318
+ # discarded the uncommitted half. False both when the snapshot succeeded (the
319
+ # dirty ref subsumes the branch) and when the tree was clean (nothing to
320
+ # capture, so the commits branch IS the whole attempt) — the ref name alone
321
+ # cannot tell those apart, which is why this is recorded rather than derived.
322
+ # Cleared with `preserve_ref`. Survives the resume serialization round-trip.
323
+ preserve_partial: bool = False
324
+ # set by runs.rearm_escalation: this task was re-armed out of ESCALATED for a
325
+ # clean rebuild against the corrected spec (not a failed attempt). Lets the
326
+ # resume-time manual-recovery notice describe the real cause; cleared once the
327
+ # rebuild proceeds. Survives the resume serialization round-trip.
328
+ rearmed: bool = False
329
+ # latched True for the lifetime of a resolved-escalation re-drive (set when
330
+ # _finish_inflight re-drives a `rearmed` task, cleared once the corrected spec
331
+ # is committed). While set, every rollback preserves the FROID artifact folders'
332
+ # tracked content, so a mid-re-drive retry/defer reset can't silently revert
333
+ # the human correction. Survives the resume serialization round-trip.
334
+ resolved_redrive: bool = False
335
+ # stories mode only: set when a spec_checkpoint story's plan-halt leg verified
336
+ # (spec at ready-for-dev) and the run paused for human plan review. On resume
337
+ # StoriesEngine._resume_after_dev_verify reads it to re-drive the implement leg
338
+ # (rather than the base review+commit) and clears it. Survives the round-trip.
339
+ plan_checkpoint_pending: bool = False
340
+ # stories mode only: the durable "a human plan review is still owed" obligation
341
+ # for a spec_checkpoint story. Latched at the story's first (leg-1) dispatch —
342
+ # BEFORE the session runs and keyed off the entry's spec_checkpoint flag, not
343
+ # the leg's on-disk status or result — so it survives a crash, a non-fixable
344
+ # retry, or a skill that overran `Halt after planning.`, none of which the
345
+ # on-disk-status-keyed _plan_halt_leg / result-keyed plan_checkpoint_pending
346
+ # carry across. Cleared ONLY when a plan-review pause actually raises (the
347
+ # obligation is discharged). While set after a dev leg that did not itself pause,
348
+ # StoriesEngine pauses before commit so the story can never commit un-reviewed.
349
+ plan_review_owed: bool = False
350
+ # stories mode only: the fixed slug ("unresolved" / "ambiguous") of a pre-planning
351
+ # halt sentinel this task was detected as — recorded at detection time (pick-time
352
+ # wedge or post-dev read-back), NOT re-derived from the spec_file basename at
353
+ # re-arm. runs.rearm_escalation deletes a sentinel only when this is set, so a real
354
+ # story spec that merely happens to be named `<key>-unresolved.md`, or a
355
+ # non-sentinel escalation whose spec matches the convention, is status-flipped and
356
+ # kept, never deleted. "" = not a sentinel. Survives the round-trip.
357
+ sentinel_kind: str = ""
358
+ # intent-gap patch-restore re-drive (Froid Plane #2564): a repo-relative-or-
359
+ # absolute path to the patch file froid-build-auto saved of the reverted attempt.
360
+ # Latched by runs.rearm_escalation when the human confirms the attempted reading
361
+ # was correct; the engine re-applies it onto the baseline after every reset of
362
+ # the re-drive so the re-driven session resumes review (step-04) on the restored
363
+ # diff, and clears it once the corrected work commits. None = ordinary
364
+ # from-scratch re-drive. Survives the resume serialization round-trip.
365
+ restore_patch: str | None = None
366
+ # sweep bundles only: the deferred-work ids this task closes and the
367
+ # rendered intent file handed to dev sessions
368
+ dw_ids: list[str] = field(default_factory=list)
369
+ bundle_file: str | None = None
370
+ # worktree-isolation mode only (scm.isolation = "worktree"): the unit's
371
+ # mounted worktree dir and branch, recorded so a paused/crashed run can
372
+ # reconstruct or discard the in-flight worktree on resume.
373
+ worktree_path: str = ""
374
+ branch: str = ""
375
+ sessions: list[SessionRecord] = field(default_factory=list)
376
+ tokens: TokenUsage = field(default_factory=TokenUsage)
377
+ # latched the first time this story's cost-weighted spend crossed
378
+ # limits.max_tokens_per_story at a session boundary, so the advisory notice
379
+ # fires once per STORY rather than once per session after the crossing
380
+ # (every later session of an overrunning story is over the cap too). Never
381
+ # cleared: the crossing is a fact about the story's spend, and the raw
382
+ # counts it was computed from stay in `tokens`. Persisted precisely so a
383
+ # resumed run does not re-notify what the pre-pause process already did.
384
+ token_budget_warned: bool = False
385
+
386
+ @property
387
+ def terminal(self) -> bool:
388
+ return self.phase in TERMINAL_PHASES
389
+
390
+ def record_session(self, record: SessionRecord) -> None:
391
+ self.sessions.append(record)
392
+ if record.usage:
393
+ self.tokens.add(record.usage)
394
+
395
+ def attach_session_usage(self, task_id: str, usage: TokenUsage | None) -> None:
396
+ """Fold usage into the most recent session for `task_id`. Usage is
397
+ best-effort metadata attached after the session itself is saved, so a
398
+ failed usage read never costs the recorded session."""
399
+ if usage is None:
400
+ return
401
+ for record in reversed(self.sessions):
402
+ if record.task_id != task_id:
403
+ continue
404
+ if record.usage is None:
405
+ record.usage = usage
406
+ self.tokens.add(usage)
407
+ return
408
+ raise KeyError(task_id)
409
+
410
+ def to_dict(self) -> dict[str, Any]:
411
+ return {
412
+ "story_key": self.story_key,
413
+ "epic": self.epic,
414
+ "phase": str(self.phase),
415
+ "attempt": self.attempt,
416
+ "review_cycle": self.review_cycle,
417
+ "followup_reviews_spent": self.followup_reviews_spent,
418
+ "generation": self.generation,
419
+ "followup_review_recommended": self.followup_review_recommended,
420
+ "baseline_commit": self.baseline_commit,
421
+ "baseline_untracked": self.baseline_untracked,
422
+ "baseline_ledger_digest": self.baseline_ledger_digest,
423
+ "pre_harvest_ledger": self.pre_harvest_ledger,
424
+ "pre_harvest_ledger_captured": self.pre_harvest_ledger_captured,
425
+ "post_engine_ledger_digest": self.post_engine_ledger_digest,
426
+ "harvest_wrote_ledger": self.harvest_wrote_ledger,
427
+ "ledger_changed_before_harvest": self.ledger_changed_before_harvest,
428
+ "harvested_deferrals": self.harvested_deferrals,
429
+ "bundle_closes_intended": self.bundle_closes_intended,
430
+ "refiled_followups": self.refiled_followups,
431
+ "story_closes_intended": self.story_closes_intended,
432
+ "board_advance_intended": self.board_advance_intended,
433
+ "accepted_dev_session_index": self.accepted_dev_session_index,
434
+ "harvest_carry_commit_pending": self.harvest_carry_commit_pending,
435
+ "isolated_ledger_carried": self.isolated_ledger_carried,
436
+ "spec_file": self._serialized_worktree_path(self.spec_file),
437
+ "dispatched_spec_file": self._serialized_worktree_path(self.dispatched_spec_file),
438
+ "dispatched_spec_snapshot": (
439
+ base64.b64encode(self.dispatched_spec_snapshot).decode("ascii")
440
+ if self.dispatched_spec_snapshot is not None
441
+ else None
442
+ ),
443
+ "commit_sha": self.commit_sha,
444
+ "operator_actions": self.operator_actions,
445
+ "defer_reason": self.defer_reason,
446
+ "preserve_ref": self.preserve_ref,
447
+ "preserve_partial": self.preserve_partial,
448
+ "rearmed": self.rearmed,
449
+ "resolved_redrive": self.resolved_redrive,
450
+ "plan_checkpoint_pending": self.plan_checkpoint_pending,
451
+ "plan_review_owed": self.plan_review_owed,
452
+ "sentinel_kind": self.sentinel_kind,
453
+ "restore_patch": self.restore_patch,
454
+ "dw_ids": self.dw_ids,
455
+ "bundle_file": self.bundle_file,
456
+ "worktree_path": self.worktree_path,
457
+ "branch": self.branch,
458
+ "sessions": [s.to_dict() for s in self.sessions],
459
+ "tokens": self.tokens.to_dict(),
460
+ "token_budget_warned": self.token_budget_warned,
461
+ }
462
+
463
+ def _serialized_worktree_path(self, path: str | None) -> str | None:
464
+ """Persist a worktree-local spec path relative to its mounted root.
465
+
466
+ Both the accepted/result spec and the attempt-owned dispatched spec use
467
+ this one normalization path so their state.json representations cannot
468
+ drift. In-place and outside-worktree paths remain verbatim.
469
+ """
470
+ if not path or not self.worktree_path:
471
+ return path
472
+ try:
473
+ # as_posix: persist the relative path with forward slashes so state.json
474
+ # stays portable across OSes (matches the in-worktree spec layout).
475
+ return Path(path).relative_to(self.worktree_path).as_posix()
476
+ except ValueError:
477
+ return path # spec lives outside the worktree; keep absolute
478
+
479
+ def release_spec_paths_from_mount(self) -> None:
480
+ """Give up the spec ownership a mount being DISCARDED carried.
481
+
482
+ The counterpart to :meth:`rebase_spec_paths_on`, and deliberately not its
483
+ exact inverse — the two fields part company here because their roles do:
484
+
485
+ * `dispatched_spec_file` / `dispatched_spec_snapshot` are the ATTEMPT's
486
+ binding, the pair `recovery_flow` restores bytes through. The attempt died
487
+ with its tree, so the binding has nothing left to name; clearing both
488
+ together keeps the authority pair whole (a path without its snapshot is the
489
+ one shape `_bind_dispatched_spec_for_attempt` never persists).
490
+ * `spec_file` is the ACCEPTED artifact and outlives the attempt. The
491
+ replacement mount will carry the same story's spec at the same
492
+ mount-relative place, so the relative spelling is the one that re-resolves
493
+ onto it — `verify.resolve_spec_path` probes a relative value against the
494
+ live workspace and passes an absolute one through untouched.
495
+
496
+ Leaving `spec_file` absolute into the deleted mount is what made the fresh
497
+ attempt start UNBOUND: `_dispatched_spec_for_attempt` resolves it
498
+ `strict=True`, the dead path raises, and the miss is silent because an
499
+ unbound attempt is a legal state. `_record_dev_spec` cannot repair it either
500
+ — it no-ops while `spec_file` is set.
501
+
502
+ Uses the same relativization as `to_dict`, so the discarded-mount spelling
503
+ and the persisted one cannot drift, which also means a spec OUTSIDE the mount
504
+ stays verbatim: it was never the mount's to give up. MUST be called while
505
+ `worktree_path` still names the mount.
506
+ """
507
+ self.dispatched_spec_file = None
508
+ self.dispatched_spec_snapshot = None
509
+ self.spec_file = self._serialized_worktree_path(self.spec_file)
510
+
511
+ def release_mount_owned_state(self) -> None:
512
+ """Give up EVERYTHING a mount owned: its spec ownership and the measurements
513
+ taken inside it.
514
+
515
+ One method because the two callers that stop using a mount — the restart
516
+ discard and the isolation-flip arm — must give up the same set, and the second
517
+ was written releasing only the spec half. That half-release is not a smaller
518
+ version of the same thing, it is a different bug: `baseline_commit` and
519
+ `baseline_untracked` are stamped from `self.workspace.root` (the unit under
520
+ isolation), so leaving them set hands unit-mount operands to
521
+ `recovery_flow.rollback_or_pause` running against the MAIN checkout. Neither
522
+ fails loud there — linked worktrees share the object database, so the baseline
523
+ still resolves and a reset onto it succeeds, while a fresh worktree is a
524
+ tracked-only checkout whose empty untracked snapshot makes
525
+ `verify._rollback_cleanup_plan` compute `untracked_files(repo) -
526
+ baseline_untracked` as every untracked file in the operator's own checkout.
527
+ Under an auto-recovering cause those are DELETED.
528
+
529
+ Costs the re-run nothing: `_dev_phase` re-stamps both from whatever workspace
530
+ it re-enters with, so clearing turns the `baseline_commit` leg into a correct
531
+ no-op instead of a probe of the wrong tree.
532
+
533
+ MUST be called while `worktree_path` still names the mount — the spec
534
+ relativization is measured against it.
535
+ """
536
+ self.release_spec_paths_from_mount()
537
+ self.baseline_commit = None
538
+ self.baseline_untracked = None
539
+
540
+ def rebase_spec_paths_on(self, root: Path) -> None:
541
+ """Re-absolutize both spec-ownership paths against the tree that owns them.
542
+
543
+ The read-side inverse of :meth:`_serialized_worktree_path`, and the single
544
+ implementation of that rule: `to_dict` persists a worktree-local spec
545
+ RELATIVE to the mount and `from_dict` reads it back raw, so a consumer that
546
+ resolves the raw value against anything else names the wrong tree. The main
547
+ checkout carries the same `_froid-output/...` layout, so that wrong tree
548
+ answers `is_file()` and passes containment — the failure is silent, not an
549
+ error.
550
+
551
+ Both fields move together because they are one asymmetry: `spec_file` is the
552
+ accepted/result artifact and `dispatched_spec_file` the attempt-owned input,
553
+ and a caller re-anchoring one and not the other leaves a task naming two
554
+ trees at once.
555
+
556
+ Idempotent: an absolute value is already anchored (a spec outside the mount
557
+ is persisted verbatim) and passes through untouched, so re-running this
558
+ against the same root cannot double-join. `root` is the tree the values were
559
+ persisted relative to — `task.worktree_path` — never the caller's cwd or
560
+ project.
561
+ """
562
+ self.spec_file = _rebased_on(self.spec_file, root)
563
+ self.dispatched_spec_file = _rebased_on(self.dispatched_spec_file, root)
564
+
565
+ @classmethod
566
+ def from_dict(cls, d: dict[str, Any]) -> "StoryTask":
567
+ dispatched_spec_snapshot = d.get("dispatched_spec_snapshot")
568
+ if dispatched_spec_snapshot is not None:
569
+ try:
570
+ dispatched_spec_snapshot = base64.b64decode(
571
+ str(dispatched_spec_snapshot).encode("ascii"),
572
+ validate=True,
573
+ )
574
+ except ValueError as exc:
575
+ raise ValueError(
576
+ f"story {d.get('story_key')!r}: dispatched_spec_snapshot " "is not valid base64"
577
+ ) from exc
578
+ return cls(
579
+ story_key=d["story_key"],
580
+ epic=int(d["epic"]),
581
+ phase=Phase(d["phase"]),
582
+ attempt=int(d.get("attempt", 0)),
583
+ review_cycle=int(d.get("review_cycle", 0)),
584
+ followup_reviews_spent=int(d.get("followup_reviews_spent", 0)),
585
+ generation=int(d.get("generation", 0)),
586
+ followup_review_recommended=bool(d.get("followup_review_recommended", False)),
587
+ baseline_commit=d.get("baseline_commit"),
588
+ baseline_untracked=(
589
+ [str(p) for p in d["baseline_untracked"]]
590
+ if d.get("baseline_untracked") is not None
591
+ else None
592
+ ),
593
+ baseline_ledger_digest=(
594
+ str(d.get("baseline_ledger_digest"))
595
+ if d.get("baseline_ledger_digest") is not None
596
+ else None
597
+ ),
598
+ pre_harvest_ledger=(
599
+ str(d.get("pre_harvest_ledger"))
600
+ if d.get("pre_harvest_ledger") is not None
601
+ else None
602
+ ),
603
+ pre_harvest_ledger_captured=bool(d.get("pre_harvest_ledger_captured", False)),
604
+ post_engine_ledger_digest=(
605
+ str(d.get("post_engine_ledger_digest"))
606
+ if d.get("post_engine_ledger_digest") is not None
607
+ else None
608
+ ),
609
+ harvest_wrote_ledger=bool(d.get("harvest_wrote_ledger", False)),
610
+ ledger_changed_before_harvest=bool(d.get("ledger_changed_before_harvest", False)),
611
+ harvested_deferrals=[deepcopy(dict(item)) for item in d.get("harvested_deferrals", [])],
612
+ bundle_closes_intended=[str(i) for i in d.get("bundle_closes_intended", [])],
613
+ refiled_followups=[deepcopy(dict(item)) for item in d.get("refiled_followups", [])],
614
+ story_closes_intended=[str(i) for i in d.get("story_closes_intended", [])],
615
+ board_advance_intended=(
616
+ str(d["board_advance_intended"])
617
+ if d.get("board_advance_intended") is not None
618
+ else None
619
+ ),
620
+ accepted_dev_session_index=(
621
+ int(d["accepted_dev_session_index"])
622
+ if d.get("accepted_dev_session_index") is not None
623
+ else None
624
+ ),
625
+ harvest_carry_commit_pending=bool(d.get("harvest_carry_commit_pending", False)),
626
+ isolated_ledger_carried=bool(d.get("isolated_ledger_carried", False)),
627
+ spec_file=d.get("spec_file"),
628
+ dispatched_spec_file=d.get("dispatched_spec_file"),
629
+ dispatched_spec_snapshot=dispatched_spec_snapshot,
630
+ commit_sha=d.get("commit_sha"),
631
+ operator_actions=[str(a) for a in d.get("operator_actions", [])],
632
+ defer_reason=d.get("defer_reason"),
633
+ preserve_ref=d.get("preserve_ref"),
634
+ preserve_partial=bool(d.get("preserve_partial", False)),
635
+ rearmed=bool(d.get("rearmed", False)),
636
+ resolved_redrive=bool(d.get("resolved_redrive", False)),
637
+ plan_checkpoint_pending=bool(d.get("plan_checkpoint_pending", False)),
638
+ plan_review_owed=bool(d.get("plan_review_owed", False)),
639
+ sentinel_kind=str(d.get("sentinel_kind", "")),
640
+ restore_patch=d.get("restore_patch"),
641
+ dw_ids=[str(i) for i in d.get("dw_ids", [])],
642
+ bundle_file=d.get("bundle_file"),
643
+ worktree_path=str(d.get("worktree_path", "")),
644
+ branch=str(d.get("branch", "")),
645
+ sessions=[SessionRecord.from_dict(s) for s in d.get("sessions", [])],
646
+ tokens=TokenUsage.from_dict(d.get("tokens", {})),
647
+ token_budget_warned=bool(d.get("token_budget_warned", False)),
648
+ )
649
+
650
+
651
+ @dataclass
652
+ class RunState:
653
+ run_id: str
654
+ project: str
655
+ started_at: str
656
+ # The git root this run's code work happens in — `paths.repo_root`, which is
657
+ # `paths.project` unless `_froid/bmm/config.yaml` sets a `repo_root:` override.
658
+ # Persisted because `runs.rearm_escalation` runs OUT OF PROCESS from the engine
659
+ # and had only `project` to reach for, so it advanced the attempt baseline by
660
+ # reading HEAD of a repo the proof-of-work gate never measures. Empty means a
661
+ # state.json written before this field existed; `code_root` then falls back to
662
+ # `project`, which is exactly the pre-upgrade behavior and the correct answer
663
+ # for every run without the override.
664
+ repo_root: str = ""
665
+ policy_snapshot: dict[str, Any] = field(default_factory=dict)
666
+ # SECONDARY copy of the host-exec baseline (#498) — runsetup.config_digest over
667
+ # the agent-writable config that reaches HOST code execution: verify commands,
668
+ # the resolved launch binary/args/env, the plugin allowlist (#461 point 4).
669
+ #
670
+ # The one resume TRUSTS is out of the tree (`runs.write_trusted_config_digest`),
671
+ # because a baseline whose whole job is to police the agent-writable tree cannot
672
+ # live in it — a session that rewrote policy.toml could blank this field in the
673
+ # same breath and silence the warning `resume` owes the operator. So this copy is
674
+ # never preferred: `_resume_paused_run` consults it ONLY when the state root
675
+ # holds no file for the run, which is what keeps rewriting it pointless (the
676
+ # #498 attack test asserts exactly that).
677
+ #
678
+ # It is still written, and must be, for the two cases where the out-of-tree file
679
+ # is honestly absent rather than tampered away — in both, this copy is the run's
680
+ # only surviving pin:
681
+ # * the state root is keyed by the project's RESOLVED PATH (`runs.project_tag`),
682
+ # so moving or renaming the project keys the run somewhere new and orphans
683
+ # its state subtree (FEATURES.md documents the GC half of this). state.json
684
+ # lives in the run dir and travels with it. Same for a FROID_LOOP_STATE_DIR
685
+ # that changes between launch and resume.
686
+ # * a run PAUSED before #498 has its baseline here and nowhere else; the first
687
+ # resume under this code reads it and mints the out-of-tree file.
688
+ # Dropping it would turn "this run has a pin" into "this run has none" in all of
689
+ # them. Empty means what it always did — no prior pin, hence no warning — which
690
+ # is why the resume compare is guarded on non-emptiness. The auto-sweep gate has
691
+ # never read this field: it compares against its own in-memory closure baseline,
692
+ # which no session can reach.
693
+ trusted_config_digest: str = ""
694
+ current_epic: int | None = None
695
+ # the run's story scope + cap, as passed on the launching CLI (`--epic`,
696
+ # `--story`, `--max-stories`). Persisted so `resume` rebuilds the Engine with
697
+ # the SAME selector — otherwise a resumed `--epic N` run silently widens to
698
+ # every epic and can jump out of its scope at the next pick.
699
+ epic_filter: int | None = None
700
+ story_filter: str | None = None
701
+ max_stories: int | None = None
702
+ paused_reason: str | None = None
703
+ paused_stage: str | None = None
704
+ paused_story_key: str | None = None
705
+ finished: bool = False
706
+ # deliberately stopped (froid-loop stop / engine SIGTERM); distinct from a
707
+ # crash. Resume clears it via clear_pause(), so a stopped run is resumable.
708
+ stopped: bool = False
709
+ # an unexpected exception escaped Engine.run() and was recorded (crash.txt +
710
+ # run-crash journal). Distinct from `stopped`; resume clears it via
711
+ # clear_pause() so a crashed run re-arms like a stopped one. crash_error is a
712
+ # short "Type: message" for display; the full traceback lives in crash.txt.
713
+ crashed: bool = False
714
+ crash_error: str | None = None
715
+ run_type: str = "story" # "story" | "sweep" — resume/status dispatch on it
716
+ # story-queue source (policy.StoriesPolicy.source), pinned at run start so
717
+ # resume/resolve rebuild the right engine (StoriesEngine vs the sprint Engine)
718
+ # without re-reading policy — a policy edit mid-run must not switch a live run's
719
+ # mode. `run_type` stays "story" for both; `source` selects the picker.
720
+ source: str = "sprint-status"
721
+ # stories mode only: the project-relative (or absolute) spec folder holding
722
+ # stories.yaml + SPEC.md. Empty under sprint-status.
723
+ spec_folder: str = ""
724
+ # sweep runs only: the triage->bundles cycle in progress; 1 maps to the
725
+ # legacy (unsuffixed) artifact names so old paused runs resume unchanged
726
+ sweep_cycle: int = 1
727
+ # auto-sweep triggers already fired this run (e.g. "epic-1", "run-end");
728
+ # guards re-fire on resume
729
+ sweeps_triggered: list[str] = field(default_factory=list)
730
+ # auto-sweep triggers this run did NOT deliver, trigger -> SWEEP_REFUSED_*.
731
+ # Kept apart from sweeps_triggered rather than folded into it: that list is
732
+ # the re-fire latch, and widening it to a mapping would silently degrade the
733
+ # per-element sanitizer loop in diagnostics.py. A trigger may appear in both
734
+ # (SWEEP_REFUSED_FAILED = a child that started and then failed).
735
+ sweeps_refused: dict[str, str] = field(default_factory=dict)
736
+ # worktree-isolation mode only: the branch every unit merges back into,
737
+ # resolved once at run start (default = the branch checked out then) and
738
+ # pinned so resume keeps targeting the same branch.
739
+ target_branch: str = ""
740
+ # free-form scratch space shared across plugin hooks (HookContext.shared).
741
+ # Persisted so a plugin's cross-stage state survives pause/resume; values
742
+ # MUST be JSON-serializable. Empty + untouched on a zero-plugin run.
743
+ plugin_shared: dict[str, Any] = field(default_factory=dict)
744
+ tasks: dict[str, StoryTask] = field(default_factory=dict)
745
+
746
+ @property
747
+ def paused(self) -> bool:
748
+ return self.paused_reason is not None
749
+
750
+ @property
751
+ def code_root(self) -> Path:
752
+ """The tree git runs against for this run — ``repo_root`` when the run
753
+ recorded one, else ``project``.
754
+
755
+ The single reader of the pair, so an out-of-process consumer
756
+ (``runs.rearm_escalation``) cannot pick the wrong one, and a pre-upgrade
757
+ state.json (empty ``repo_root``) degrades to precisely what it did before
758
+ rather than to a path that does not exist."""
759
+ return Path(self.repo_root or self.project)
760
+
761
+ def handled_keys(self) -> set[str]:
762
+ """Story keys this run already drove to a terminal phase."""
763
+ return {k for k, t in self.tasks.items() if t.terminal}
764
+
765
+ def clear_pause(self) -> None:
766
+ self.paused_reason = None
767
+ self.paused_stage = None
768
+ self.paused_story_key = None
769
+ self.stopped = False
770
+ self.crashed = False
771
+ self.crash_error = None
772
+
773
+ def cache_read_weight(self) -> float:
774
+ """The run's cache-read weight from its persisted policy snapshot; the
775
+ product default (policy.LimitsPolicy.cache_read_weight = 0.1) when the
776
+ snapshot predates the field or is malformed. Lets the TUI show the same
777
+ weighted total the engine's budget uses without importing Policy.
778
+
779
+ The snapshot is re-stamped at every engine start (run, sweep, resume), so
780
+ on a resumed run this is the *resuming* process's weight, matching what
781
+ that process enforces. Edit the weight and resume and the run's whole
782
+ accumulated history re-weights — totals are recomputed from raw counts,
783
+ and the budget has always judged cumulative counts at the live weight."""
784
+ limits = self.policy_snapshot.get("limits")
785
+ if isinstance(limits, dict):
786
+ try:
787
+ return float(limits["cache_read_weight"])
788
+ except (KeyError, TypeError, ValueError):
789
+ pass
790
+ return 0.1
791
+
792
+ def to_dict(self) -> dict[str, Any]:
793
+ return {
794
+ "run_id": self.run_id,
795
+ "project": self.project,
796
+ "repo_root": self.repo_root,
797
+ "started_at": self.started_at,
798
+ "policy_snapshot": self.policy_snapshot,
799
+ "trusted_config_digest": self.trusted_config_digest,
800
+ "current_epic": self.current_epic,
801
+ "epic_filter": self.epic_filter,
802
+ "story_filter": self.story_filter,
803
+ "max_stories": self.max_stories,
804
+ "paused_reason": self.paused_reason,
805
+ "paused_stage": self.paused_stage,
806
+ "paused_story_key": self.paused_story_key,
807
+ "finished": self.finished,
808
+ "stopped": self.stopped,
809
+ "crashed": self.crashed,
810
+ "crash_error": self.crash_error,
811
+ "run_type": self.run_type,
812
+ "source": self.source,
813
+ "spec_folder": self.spec_folder,
814
+ "sweep_cycle": self.sweep_cycle,
815
+ "sweeps_triggered": self.sweeps_triggered,
816
+ "sweeps_refused": self.sweeps_refused,
817
+ "target_branch": self.target_branch,
818
+ "plugin_shared": self.plugin_shared,
819
+ "tasks": {k: t.to_dict() for k, t in self.tasks.items()},
820
+ }
821
+
822
+ @classmethod
823
+ def from_dict(cls, d: dict[str, Any]) -> "RunState":
824
+ return cls(
825
+ run_id=d["run_id"],
826
+ project=d["project"],
827
+ repo_root=str(d.get("repo_root", "")),
828
+ started_at=d["started_at"],
829
+ policy_snapshot=d.get("policy_snapshot", {}),
830
+ trusted_config_digest=str(d.get("trusted_config_digest", "")),
831
+ current_epic=d.get("current_epic"),
832
+ epic_filter=d.get("epic_filter"),
833
+ story_filter=d.get("story_filter"),
834
+ max_stories=d.get("max_stories"),
835
+ paused_reason=d.get("paused_reason"),
836
+ paused_stage=d.get("paused_stage"),
837
+ paused_story_key=d.get("paused_story_key"),
838
+ finished=bool(d.get("finished", False)),
839
+ stopped=bool(d.get("stopped", False)),
840
+ crashed=bool(d.get("crashed", False)),
841
+ crash_error=d.get("crash_error"),
842
+ run_type=str(d.get("run_type", "story")),
843
+ source=str(d.get("source", "sprint-status")),
844
+ spec_folder=str(d.get("spec_folder", "")),
845
+ sweep_cycle=int(d.get("sweep_cycle", 1)),
846
+ sweeps_triggered=[str(s) for s in d.get("sweeps_triggered", [])],
847
+ sweeps_refused={str(k): str(v) for k, v in d.get("sweeps_refused", {}).items()},
848
+ target_branch=str(d.get("target_branch", "")),
849
+ plugin_shared=dict(d.get("plugin_shared", {})),
850
+ tasks={k: StoryTask.from_dict(t) for k, t in d.get("tasks", {}).items()},
851
+ )
852
+
853
+
854
+ @dataclass(frozen=True)
855
+ class VerifyOutcome:
856
+ ok: bool
857
+ reason: str = ""
858
+ severity: str = "" # "" | "CRITICAL" | "PREFERENCE" — set when not retryable
859
+ # fixable failures carry concrete evidence (failing command output) that a
860
+ # feedback-driven repair session can act on; non-fixable retries start over
861
+ fixable: bool = False
862
+ # the failure is the run environment's, not the story's (verify command
863
+ # not found / not executable): no repair session can fix it and every
864
+ # story shares the same commands, so it must never charge attempt budgets
865
+ env_fault: bool = False
866
+ # a session deliberately contradicted a state the orchestrator had already
867
+ # established (a review revoking the sprint sign-off it advanced at dev
868
+ # time): no further session can reconcile it, so it routes to a pause with
869
+ # both sides named rather than to another cycle (#334)
870
+ contradiction: bool = False
871
+
872
+ @classmethod
873
+ def passed(cls) -> "VerifyOutcome":
874
+ return cls(ok=True)
875
+
876
+ @classmethod
877
+ def retry(cls, reason: str, fixable: bool = False) -> "VerifyOutcome":
878
+ return cls(ok=False, reason=reason, fixable=fixable)
879
+
880
+ @classmethod
881
+ def escalate(
882
+ cls,
883
+ reason: str,
884
+ severity: str = "CRITICAL",
885
+ env_fault: bool = False,
886
+ contradiction: bool = False,
887
+ ) -> "VerifyOutcome":
888
+ return cls(
889
+ ok=False,
890
+ reason=reason,
891
+ severity=severity,
892
+ env_fault=env_fault,
893
+ contradiction=contradiction,
894
+ )
895
+
896
+ @property
897
+ def retryable(self) -> bool:
898
+ return not self.ok and not self.severity