froid-loop 0.11.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- froid_loop/__init__.py +11 -0
- froid_loop/__main__.py +12 -0
- froid_loop/adapters/__init__.py +3 -0
- froid_loop/adapters/base.py +254 -0
- froid_loop/adapters/entrypoints.py +63 -0
- froid_loop/adapters/env_fault.py +290 -0
- froid_loop/adapters/generic.py +2013 -0
- froid_loop/adapters/mock.py +49 -0
- froid_loop/adapters/multiplexer.py +914 -0
- froid_loop/adapters/opencode_http.py +1687 -0
- froid_loop/adapters/profile.py +650 -0
- froid_loop/adapters/psmux_backend.py +1428 -0
- froid_loop/adapters/registry.py +322 -0
- froid_loop/adapters/tmux_backend.py +35 -0
- froid_loop/adapters/tmux_base.py +630 -0
- froid_loop/checks.py +187 -0
- froid_loop/cli.py +5041 -0
- froid_loop/data/__init__.py +0 -0
- froid_loop/data/froid_loop_hook.py +228 -0
- froid_loop/data/froid_loop_probe_hook.py +88 -0
- froid_loop/data/plugins/example/plugin.toml +21 -0
- froid_loop/data/plugins/tea/plugin.toml +184 -0
- froid_loop/data/plugins/tea/tea_plugin.py +258 -0
- froid_loop/data/plugins/unity/plugin.toml +140 -0
- froid_loop/data/plugins/unity/unity_assets/FroidLoop.Unity.Editor.asmdef +16 -0
- froid_loop/data/plugins/unity/unity_assets/FroidLoop.Unity.Editor.asmdef.meta +7 -0
- froid_loop/data/plugins/unity/unity_assets/SceneAutoSaveGuard.cs +221 -0
- froid_loop/data/plugins/unity/unity_assets/SceneAutoSaveGuard.cs.meta +11 -0
- froid_loop/data/plugins/unity/unity_assets/_folders/Editor.meta +8 -0
- froid_loop/data/plugins/unity/unity_assets/_folders/FroidLoop.meta +8 -0
- froid_loop/data/plugins/unity/unity_cleanup.py +125 -0
- froid_loop/data/plugins/unity/unity_dialog_probe.py +239 -0
- froid_loop/data/plugins/unity/unity_facts.md +17 -0
- froid_loop/data/plugins/unity/unity_plugin.py +415 -0
- froid_loop/data/plugins/unity/unity_quiesce.py +234 -0
- froid_loop/data/plugins/unity/unity_ready.py +230 -0
- froid_loop/data/plugins/unity/unity_seed_assets.py +298 -0
- froid_loop/data/plugins/unity/unity_setup.py +551 -0
- froid_loop/data/plugins/unity/unity_teardown.py +362 -0
- froid_loop/data/profiles/antigravity.toml +52 -0
- froid_loop/data/profiles/claude.toml +85 -0
- froid_loop/data/profiles/codex.toml +22 -0
- froid_loop/data/profiles/copilot.toml +52 -0
- froid_loop/data/profiles/gemini.toml +26 -0
- froid_loop/data/profiles/opencode.toml +54 -0
- froid_loop/data/settings/core.toml +458 -0
- froid_loop/data/skills/README.md +93 -0
- froid_loop/data/skills/froid-loop-resolve/SKILL.md +288 -0
- froid_loop/data/skills/froid-loop-setup/SKILL.md +161 -0
- froid_loop/data/skills/froid-loop-setup/assets/module-help.csv +3 -0
- froid_loop/data/skills/froid-loop-setup/assets/module.yaml +19 -0
- froid_loop/data/skills/froid-loop-sweep/SKILL.md +100 -0
- froid_loop/data/skills/froid-loop-sweep/automation-mode.md +127 -0
- froid_loop/data/skills/froid-loop-sweep/deferred-work-format.md +302 -0
- froid_loop/data/skills/froid-loop-sweep/migration-mode.md +86 -0
- froid_loop/decisions.py +202 -0
- froid_loop/deferredwork.py +2282 -0
- froid_loop/devcontract.py +892 -0
- froid_loop/diagnostics.py +1104 -0
- froid_loop/documents.py +532 -0
- froid_loop/engine.py +7732 -0
- froid_loop/envvars.py +111 -0
- froid_loop/escalation.py +225 -0
- froid_loop/events.py +266 -0
- froid_loop/fences.py +103 -0
- froid_loop/froidconfig.py +226 -0
- froid_loop/frontmatter.py +526 -0
- froid_loop/gates.py +133 -0
- froid_loop/install.py +2936 -0
- froid_loop/journal.py +178 -0
- froid_loop/machine.py +148 -0
- froid_loop/model.py +898 -0
- froid_loop/operatoractions.py +474 -0
- froid_loop/platform_util.py +1490 -0
- froid_loop/plugins/__init__.py +64 -0
- froid_loop/plugins/bus.py +259 -0
- froid_loop/plugins/context.py +319 -0
- froid_loop/plugins/loader.py +145 -0
- froid_loop/plugins/manifest.py +279 -0
- froid_loop/plugins/model.py +296 -0
- froid_loop/plugins/registry.py +245 -0
- froid_loop/plugins/trust.py +75 -0
- froid_loop/policy.py +1569 -0
- froid_loop/probe.py +1044 -0
- froid_loop/process_host.py +408 -0
- froid_loop/recovery_flow.py +1561 -0
- froid_loop/resolve.py +283 -0
- froid_loop/runs.py +4715 -0
- froid_loop/runsetup.py +1293 -0
- froid_loop/sanitize.py +593 -0
- froid_loop/settings_schema.py +276 -0
- froid_loop/signals.py +160 -0
- froid_loop/sprintstatus.py +609 -0
- froid_loop/statemachine.py +57 -0
- froid_loop/stories.py +615 -0
- froid_loop/stories_engine.py +796 -0
- froid_loop/sweep.py +1892 -0
- froid_loop/tokens.py +196 -0
- froid_loop/tui/__init__.py +11 -0
- froid_loop/tui/app.py +1584 -0
- froid_loop/tui/data.py +840 -0
- froid_loop/tui/launch.py +1003 -0
- froid_loop/tui/screens/__init__.py +1 -0
- froid_loop/tui/screens/dashboard.py +1071 -0
- froid_loop/tui/screens/modals.py +943 -0
- froid_loop/tui/screens/settings_screen.py +477 -0
- froid_loop/tui/settings.py +135 -0
- froid_loop/tui/widgets.py +981 -0
- froid_loop/verify.py +4545 -0
- froid_loop/workspace.py +320 -0
- froid_loop/worktree_flow.py +2301 -0
- froid_loop-0.11.1.dist-info/METADATA +728 -0
- froid_loop-0.11.1.dist-info/RECORD +116 -0
- froid_loop-0.11.1.dist-info/WHEEL +4 -0
- froid_loop-0.11.1.dist-info/entry_points.txt +2 -0
- froid_loop-0.11.1.dist-info/licenses/LICENSE +30 -0
froid_loop/model.py
ADDED
|
@@ -0,0 +1,898 @@
|
|
|
1
|
+
"""Core data model: story lifecycle phases, per-task records, run state."""
|
|
2
|
+
|
|
3
|
+
# Strict-checked under #245 Stage 2, with the two rules below relaxed for this
|
|
4
|
+
# file only. `reportUnknownVariableType`: the dataclass collection fields use the
|
|
5
|
+
# idiomatic `field(default_factory=list|dict)`, which pyright can only infer as
|
|
6
|
+
# `list[Unknown]` / `dict[Unknown, Unknown]` (it does not fold the declared
|
|
7
|
+
# annotation back into the factory) though the fields are correctly typed.
|
|
8
|
+
# `reportUnknownArgumentType`: `from_dict` / snapshot readers pull values out of
|
|
9
|
+
# run-persisted `dict[str, Any]`, so isinstance-narrowing an `Any` value yields
|
|
10
|
+
# Unknown at that boundary. Both are inherent to the persistence edge, not
|
|
11
|
+
# annotation drift; every other strict rule stays on.
|
|
12
|
+
# pyright: reportUnknownArgumentType=false, reportUnknownVariableType=false
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import base64
|
|
17
|
+
from copy import deepcopy
|
|
18
|
+
from dataclasses import dataclass, field
|
|
19
|
+
from enum import StrEnum
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
from typing import Any
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class Phase(StrEnum):
|
|
25
|
+
PENDING = "pending"
|
|
26
|
+
DEV_RUNNING = "dev-running"
|
|
27
|
+
DEV_VERIFY = "dev-verify"
|
|
28
|
+
REVIEW_RUNNING = "review-running"
|
|
29
|
+
REVIEW_VERIFY = "review-verify"
|
|
30
|
+
COMMITTING = "committing"
|
|
31
|
+
# sweep-only: the triage session classifying open deferred-work entries
|
|
32
|
+
TRIAGE_RUNNING = "triage-running"
|
|
33
|
+
TRIAGE_VERIFY = "triage-verify"
|
|
34
|
+
DONE = "done"
|
|
35
|
+
DEFERRED = "deferred"
|
|
36
|
+
ESCALATED = "escalated"
|
|
37
|
+
# the story's agent-doable work is finished and COMMITTED, but its acceptance
|
|
38
|
+
# criteria include external actions only a human can perform (buy a domain,
|
|
39
|
+
# publish a DNS record). Terminal like DONE — the run moves on rather than
|
|
40
|
+
# halting — but not DONE: the outstanding actions are recorded on the task.
|
|
41
|
+
# Deliberately NOT a pause: an operator-owed story must never block the
|
|
42
|
+
# stories behind it (contrast the spec's `Block If:` -> blocked -> CRITICAL
|
|
43
|
+
# -> pause channel, which is run-halting by design).
|
|
44
|
+
AWAITING_OPERATOR = "awaiting-operator"
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
# Terminal = the orchestrator will not drive this story further in this run. It
|
|
48
|
+
# says nothing about success: DONE and AWAITING_OPERATOR carry a commit, DEFERRED
|
|
49
|
+
# and ESCALATED do not.
|
|
50
|
+
TERMINAL_PHASES = frozenset({Phase.DONE, Phase.DEFERRED, Phase.ESCALATED, Phase.AWAITING_OPERATOR})
|
|
51
|
+
|
|
52
|
+
# Pause stages recorded in RunState.paused_stage
|
|
53
|
+
PAUSE_SPEC_APPROVAL = "spec-approval"
|
|
54
|
+
PAUSE_EPIC_BOUNDARY = "epic-boundary"
|
|
55
|
+
PAUSE_ESCALATION = "escalation"
|
|
56
|
+
# Raised by Engine._refuse_gated_story: the picked story is named by the `gate:`
|
|
57
|
+
# line of a deferred-work entry that has not landed. Produced before the story is
|
|
58
|
+
# recorded in state.tasks, so a resume re-picks it and re-reads the ledger.
|
|
59
|
+
PAUSE_STORY_GATE = "story-gate"
|
|
60
|
+
# stories-mode HITL checkpoints (independent per story). PLAN fires after a
|
|
61
|
+
# spec_checkpoint story's plan-halt leg (ready-for-dev, awaiting human plan
|
|
62
|
+
# review before implementation); STORY fires after a done_checkpoint story's
|
|
63
|
+
# commit (skip-if-last). Both re-arm through the same resume path.
|
|
64
|
+
PAUSE_PLAN_CHECKPOINT = "plan-checkpoint"
|
|
65
|
+
PAUSE_STORY_CHECKPOINT = "story-checkpoint"
|
|
66
|
+
|
|
67
|
+
# Reasons recorded in RunState.sweeps_refused (trigger -> reason). A CLOSED
|
|
68
|
+
# vocabulary of short slugs, deliberately not a formatted exception: `froid-loop
|
|
69
|
+
# diagnose` renders run state through `sanitize.guard`, which *raises*
|
|
70
|
+
# LeakDetected on a home-path hit rather than redacting it (genuine PII never
|
|
71
|
+
# auto-repairs — sanitize.py). A free-form `str(e)` here would therefore make the
|
|
72
|
+
# dump fail outright on exactly the runs worth dumping, and the per-value
|
|
73
|
+
# `looks_like_identifier` filter would blank it anyway. Add a slug, never a
|
|
74
|
+
# message.
|
|
75
|
+
SWEEP_REFUSED_NOT_STARTED = "not-started" # the launch raised before a child existed
|
|
76
|
+
SWEEP_REFUSED_FAILED = "failed" # a child started, then failed
|
|
77
|
+
SWEEP_REFUSED_DIRTY = "dirty" # the worktree was unclean, or `git status` faulted
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
@dataclass
|
|
81
|
+
class TokenUsage:
|
|
82
|
+
input_tokens: int = 0
|
|
83
|
+
output_tokens: int = 0
|
|
84
|
+
cache_read_tokens: int = 0
|
|
85
|
+
cache_creation_tokens: int = 0
|
|
86
|
+
|
|
87
|
+
def add(self, other: "TokenUsage") -> None:
|
|
88
|
+
self.input_tokens += other.input_tokens
|
|
89
|
+
self.output_tokens += other.output_tokens
|
|
90
|
+
self.cache_read_tokens += other.cache_read_tokens
|
|
91
|
+
self.cache_creation_tokens += other.cache_creation_tokens
|
|
92
|
+
|
|
93
|
+
@property
|
|
94
|
+
def total(self) -> int:
|
|
95
|
+
return (
|
|
96
|
+
self.input_tokens
|
|
97
|
+
+ self.output_tokens
|
|
98
|
+
+ self.cache_read_tokens
|
|
99
|
+
+ self.cache_creation_tokens
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
def weighted_total(self, cache_read_weight: float) -> int:
|
|
103
|
+
"""Cost-proportional total: cache reads are billed at ~0.1x base input
|
|
104
|
+
on all supported vendors (Anthropic/OpenAI/Gemini, June 2026), so raw
|
|
105
|
+
totals mostly measure context re-reads; the budget discounts them."""
|
|
106
|
+
return (
|
|
107
|
+
self.input_tokens
|
|
108
|
+
+ self.output_tokens
|
|
109
|
+
+ self.cache_creation_tokens
|
|
110
|
+
+ round(self.cache_read_tokens * cache_read_weight)
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
def to_dict(self) -> dict[str, int]:
|
|
114
|
+
return {
|
|
115
|
+
"input_tokens": self.input_tokens,
|
|
116
|
+
"output_tokens": self.output_tokens,
|
|
117
|
+
"cache_read_tokens": self.cache_read_tokens,
|
|
118
|
+
"cache_creation_tokens": self.cache_creation_tokens,
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
@classmethod
|
|
122
|
+
def from_dict(cls, d: dict[str, Any]) -> "TokenUsage":
|
|
123
|
+
return cls(
|
|
124
|
+
input_tokens=int(d.get("input_tokens", 0)),
|
|
125
|
+
output_tokens=int(d.get("output_tokens", 0)),
|
|
126
|
+
cache_read_tokens=int(d.get("cache_read_tokens", 0)),
|
|
127
|
+
cache_creation_tokens=int(d.get("cache_creation_tokens", 0)),
|
|
128
|
+
)
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
@dataclass
|
|
132
|
+
class SessionRecord:
|
|
133
|
+
task_id: str
|
|
134
|
+
role: str # "dev" | "review"
|
|
135
|
+
status: str # SessionResult.status
|
|
136
|
+
# resolved adapter identity for the session (#153 phase 1). adapter "" means
|
|
137
|
+
# the record predates identity stamping; adapter set with model "" means the
|
|
138
|
+
# session ran the CLI profile's default model (no explicit model override).
|
|
139
|
+
adapter: str = ""
|
|
140
|
+
model: str = ""
|
|
141
|
+
session_id: str | None = None
|
|
142
|
+
transcript_path: str | None = None
|
|
143
|
+
usage: TokenUsage | None = None
|
|
144
|
+
# the session's parsed result payload, persisted so a durably-saved
|
|
145
|
+
# completed session is actionable on resume, not just forensics
|
|
146
|
+
result_json: dict[str, Any] | None = None
|
|
147
|
+
|
|
148
|
+
def to_dict(self) -> dict[str, Any]:
|
|
149
|
+
return {
|
|
150
|
+
"task_id": self.task_id,
|
|
151
|
+
"role": self.role,
|
|
152
|
+
"status": self.status,
|
|
153
|
+
"adapter": self.adapter,
|
|
154
|
+
"model": self.model,
|
|
155
|
+
"session_id": self.session_id,
|
|
156
|
+
"transcript_path": self.transcript_path,
|
|
157
|
+
"usage": self.usage.to_dict() if self.usage else None,
|
|
158
|
+
"result_json": self.result_json,
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
@classmethod
|
|
162
|
+
def from_dict(cls, d: dict[str, Any]) -> "SessionRecord":
|
|
163
|
+
usage = d.get("usage")
|
|
164
|
+
return cls(
|
|
165
|
+
task_id=d["task_id"],
|
|
166
|
+
role=d["role"],
|
|
167
|
+
status=d["status"],
|
|
168
|
+
adapter=str(d.get("adapter", "")),
|
|
169
|
+
model=str(d.get("model", "")),
|
|
170
|
+
session_id=d.get("session_id"),
|
|
171
|
+
transcript_path=d.get("transcript_path"),
|
|
172
|
+
usage=TokenUsage.from_dict(usage) if usage else None,
|
|
173
|
+
result_json=d.get("result_json"),
|
|
174
|
+
)
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def _rebased_on(path: str | None, root: Path) -> str | None:
|
|
178
|
+
"""One persisted spec path, re-anchored on `root`; absolute values pass through.
|
|
179
|
+
|
|
180
|
+
Split out so `StoryTask.rebase_spec_paths_on` states the rule once per field
|
|
181
|
+
without repeating the guard, and so the guard itself is unmissable: the
|
|
182
|
+
is-absolute test is what keeps an out-of-mount spec (persisted verbatim by
|
|
183
|
+
`_serialized_worktree_path`) from being joined onto a root that does not
|
|
184
|
+
contain it.
|
|
185
|
+
"""
|
|
186
|
+
if not path or Path(path).is_absolute():
|
|
187
|
+
return path
|
|
188
|
+
return str(root / path)
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
@dataclass
|
|
192
|
+
class StoryTask:
|
|
193
|
+
story_key: str
|
|
194
|
+
epic: int
|
|
195
|
+
phase: Phase = Phase.PENDING
|
|
196
|
+
attempt: int = 0
|
|
197
|
+
review_cycle: int = 0
|
|
198
|
+
# count of review rounds granted *solely* because a completed round finalized
|
|
199
|
+
# the story (status: done) yet still set `followup_review_recommended: true`.
|
|
200
|
+
# Bounded by limits.max_followup_reviews: once spent, the next such round
|
|
201
|
+
# force-converges (verify → refile the recommendation to the ledger → commit)
|
|
202
|
+
# rather than burning another cycle. Reset to 0 by runs.rearm_escalation so a
|
|
203
|
+
# human-resolved re-drive gets a fresh damping budget. Survives the round-trip.
|
|
204
|
+
followup_reviews_spent: int = 0
|
|
205
|
+
# How many times a human re-arm (`runs.rearm_escalation`) has re-opened this
|
|
206
|
+
# task. Re-arm resets `attempt` to 0 and the next dispatch bumps it back to 1,
|
|
207
|
+
# so without a discriminator the re-minted session task_id is byte-equal to a
|
|
208
|
+
# record the ABANDONED attempt already appended to the append-only `sessions`
|
|
209
|
+
# list — and `Engine._resumable_session`, which matches on that id, replays the
|
|
210
|
+
# abandoned attempt's verdict for the fresh one (#705). Feeds
|
|
211
|
+
# `engine._session_task_id`, which emits the suffix only above zero, so every
|
|
212
|
+
# id already on disk stays byte-identical across the upgrade. `task.sessions`
|
|
213
|
+
# is deliberately NOT cleared at re-arm: the run-dir audit trail it indexes is
|
|
214
|
+
# read by a second resolve cycle.
|
|
215
|
+
generation: int = 0
|
|
216
|
+
# set from the froid-build-auto session's `followup_review_recommended`
|
|
217
|
+
# frontmatter (PR #2505): when True and review.trigger = "recommended", the
|
|
218
|
+
# orchestrator runs a follow-up review pass (froid-build-auto re-invoked on the
|
|
219
|
+
# done spec); otherwise it skips it.
|
|
220
|
+
followup_review_recommended: bool = False
|
|
221
|
+
baseline_commit: str | None = None
|
|
222
|
+
# untracked, non-ignored paths present at baseline capture (repo-relative
|
|
223
|
+
# posix). On rollback only paths NOT in this set are removed, so files the
|
|
224
|
+
# user already had on disk are never deleted. None = pre-upgrade run (no
|
|
225
|
+
# snapshot); rollback then removes no untracked files at all.
|
|
226
|
+
baseline_untracked: list[str] | None = None
|
|
227
|
+
# Deferred-work bookkeeping is persisted before its readers land so an older
|
|
228
|
+
# state.json remains resumable throughout the forward-port. The nullable
|
|
229
|
+
# snapshot text and its captured flag are deliberately separate: None means
|
|
230
|
+
# "no ledger existed", while False means "no snapshot was taken".
|
|
231
|
+
baseline_ledger_digest: str | None = None
|
|
232
|
+
pre_harvest_ledger: str | None = None
|
|
233
|
+
pre_harvest_ledger_captured: bool = False
|
|
234
|
+
# Digest of the last ledger state THIS engine left on disk: the snapshot's
|
|
235
|
+
# own text at capture, refreshed to the post-append bytes once the harvest
|
|
236
|
+
# writes. It is the compare-and-set anchor `_restore_ledger` uses to tell
|
|
237
|
+
# its own retractable write from a concurrent writer's (#286), which is why
|
|
238
|
+
# it is persisted rather than kept in memory: a crash replay must be able to
|
|
239
|
+
# recognize the dead attempt's append still sitting on disk.
|
|
240
|
+
post_engine_ledger_digest: str | None = None
|
|
241
|
+
harvest_wrote_ledger: bool = False
|
|
242
|
+
ledger_changed_before_harvest: bool = False
|
|
243
|
+
# JSON-native containers only; callers persist these through state.json.
|
|
244
|
+
harvested_deferrals: list[dict[str, Any]] = field(default_factory=list)
|
|
245
|
+
bundle_closes_intended: list[str] = field(default_factory=list)
|
|
246
|
+
# `append_entry` kwargs for review-budget follow-ups this task filed into the
|
|
247
|
+
# ACTIVE workspace's ledger, which under isolation is the unit worktree's.
|
|
248
|
+
refiled_followups: list[dict[str, Any]] = field(default_factory=list)
|
|
249
|
+
# Deferred-work ids a story DECLARED it closes (`closes_deferred:`), recorded at
|
|
250
|
+
# the commit boundary. Same ledger, same isolation problem: a gitignored path
|
|
251
|
+
# never merges out of the unit worktree, so the flip has to be re-applied.
|
|
252
|
+
story_closes_intended: list[str] = field(default_factory=list)
|
|
253
|
+
# The sprint-status stage `_post_dev_state_sync` REQUESTED for this story, or
|
|
254
|
+
# None when it never ran (sweep bundles, stories mode, the legacy path). Same
|
|
255
|
+
# isolation problem as the ledger payloads above, one file over: under
|
|
256
|
+
# `isolation = "worktree"` that advance lands on the unit worktree's board,
|
|
257
|
+
# which for a gitignored board is a seeded copy shielded from the unit commit,
|
|
258
|
+
# so the post-merge carry has to re-apply it. Latest-wins — a scalar, not a
|
|
259
|
+
# list, because the board holds one stage per story and only the accepted
|
|
260
|
+
# attempt's stage can ever be carried.
|
|
261
|
+
board_advance_intended: str | None = None
|
|
262
|
+
# Index of the append-only primary dev SessionRecord whose initial decision or
|
|
263
|
+
# later verify-repair result durably returned PROCEED. Attempt numbers can be
|
|
264
|
+
# reused after a human re-arm, so the exact record occurrence is the acceptance
|
|
265
|
+
# identity; None is legacy/unarmed.
|
|
266
|
+
accepted_dev_session_index: int | None = None
|
|
267
|
+
harvest_carry_commit_pending: bool = False
|
|
268
|
+
isolated_ledger_carried: bool = False
|
|
269
|
+
spec_file: str | None = None
|
|
270
|
+
# The spec owned by the current/last dispatched dev attempt. Unlike
|
|
271
|
+
# ``spec_file`` (the accepted/result artifact), this is bound before launch
|
|
272
|
+
# so recovery can identify an attempt's lifecycle-only residue after a crash.
|
|
273
|
+
dispatched_spec_file: str | None = None
|
|
274
|
+
# Byte-exact input contents of ``dispatched_spec_file`` for the current retry
|
|
275
|
+
# chain. The JSON representation is base64, so CRLF and non-UTF-8 bytes survive
|
|
276
|
+
# a crash/resume round-trip. The chain's first bound input is retained across
|
|
277
|
+
# fixable repairs, then restored before a fresh-baseline retry; in a resolved
|
|
278
|
+
# re-drive this is the operator-corrected spec, so child-authored body edits can
|
|
279
|
+
# never become the retained correction. Cleared after successful commit. None =
|
|
280
|
+
# unbound attempt, retired chain, or legacy state.
|
|
281
|
+
dispatched_spec_snapshot: bytes | None = None
|
|
282
|
+
commit_sha: str | None = None
|
|
283
|
+
# the external, human-only actions this story still owes when it parks at
|
|
284
|
+
# Phase.AWAITING_OPERATOR — one free-text instruction per entry, as the dev
|
|
285
|
+
# session enumerated them in the spec's `operator_actions:` frontmatter.
|
|
286
|
+
# Plain strings in v1: a per-action deterministic `check:` command is a
|
|
287
|
+
# deliberate v2 question, and strings-now/objects-later is the cheaper
|
|
288
|
+
# migration than the reverse. Empty on every other phase: owing at least one
|
|
289
|
+
# action is what selects AWAITING_OPERATOR over DONE when the park path picks
|
|
290
|
+
# a committing story's final phase. That choice is the only enforcement —
|
|
291
|
+
# nothing validates this field on load or on write, so a task carrying the
|
|
292
|
+
# phase with no actions is unreachable rather than rejected. Survives the
|
|
293
|
+
# resume serialization round-trip (it is the durable record of what the human
|
|
294
|
+
# owes, and nothing re-derives it once the session that wrote the spec is
|
|
295
|
+
# gone).
|
|
296
|
+
operator_actions: list[str] = field(default_factory=list)
|
|
297
|
+
defer_reason: str | None = None
|
|
298
|
+
# the recovery ref this attempt's work was parked on by the last auto-rollback
|
|
299
|
+
# — an `attempt-preserve/*` branch (commits above baseline) or, when the tree
|
|
300
|
+
# was also dirty, the `refs/attempt-preserve-dirty/*` snapshot, which is
|
|
301
|
+
# parented at the attempt's HEAD and therefore subsumes the branch (last
|
|
302
|
+
# writer wins, so one `git merge --ff-only <ref>` recovers the whole attempt
|
|
303
|
+
# — unless `preserve_partial` is set). Set by RecoveryFlow, cleared at the top
|
|
304
|
+
# of every auto-rollback so it can never name a *previous* attempt's ref; read
|
|
305
|
+
# by `_defer` (notification) and projected into `status`. None = the last
|
|
306
|
+
# auto-rollback parked nothing (no commits above baseline and a clean or
|
|
307
|
+
# uncapturable tree, or the ref failed to take). Isolation-INDEPENDENT: a unit
|
|
308
|
+
# worktree's own dev-retry rollback parks on the same shared refs, so a
|
|
309
|
+
# deferred isolated unit can carry BOTH a kept-failed branch (the final
|
|
310
|
+
# attempt) and a preserve_ref (an earlier, rolled-back one) — `_defer` names
|
|
311
|
+
# both. The unit branch itself is never written here: a live branch is not a
|
|
312
|
+
# parked snapshot. Not cleared on success — a mid-retry rollback's breadcrumb
|
|
313
|
+
# stays readable. Survives the resume serialization round-trip.
|
|
314
|
+
preserve_ref: str | None = None
|
|
315
|
+
# set when the auto-rollback's *worktree* snapshot was attempted and raised
|
|
316
|
+
# (journalled `attempt-worktree-preserve-failed`), so `preserve_ref` names an
|
|
317
|
+
# `attempt-preserve/*` commits branch ALONE and the reset that followed
|
|
318
|
+
# discarded the uncommitted half. False both when the snapshot succeeded (the
|
|
319
|
+
# dirty ref subsumes the branch) and when the tree was clean (nothing to
|
|
320
|
+
# capture, so the commits branch IS the whole attempt) — the ref name alone
|
|
321
|
+
# cannot tell those apart, which is why this is recorded rather than derived.
|
|
322
|
+
# Cleared with `preserve_ref`. Survives the resume serialization round-trip.
|
|
323
|
+
preserve_partial: bool = False
|
|
324
|
+
# set by runs.rearm_escalation: this task was re-armed out of ESCALATED for a
|
|
325
|
+
# clean rebuild against the corrected spec (not a failed attempt). Lets the
|
|
326
|
+
# resume-time manual-recovery notice describe the real cause; cleared once the
|
|
327
|
+
# rebuild proceeds. Survives the resume serialization round-trip.
|
|
328
|
+
rearmed: bool = False
|
|
329
|
+
# latched True for the lifetime of a resolved-escalation re-drive (set when
|
|
330
|
+
# _finish_inflight re-drives a `rearmed` task, cleared once the corrected spec
|
|
331
|
+
# is committed). While set, every rollback preserves the FROID artifact folders'
|
|
332
|
+
# tracked content, so a mid-re-drive retry/defer reset can't silently revert
|
|
333
|
+
# the human correction. Survives the resume serialization round-trip.
|
|
334
|
+
resolved_redrive: bool = False
|
|
335
|
+
# stories mode only: set when a spec_checkpoint story's plan-halt leg verified
|
|
336
|
+
# (spec at ready-for-dev) and the run paused for human plan review. On resume
|
|
337
|
+
# StoriesEngine._resume_after_dev_verify reads it to re-drive the implement leg
|
|
338
|
+
# (rather than the base review+commit) and clears it. Survives the round-trip.
|
|
339
|
+
plan_checkpoint_pending: bool = False
|
|
340
|
+
# stories mode only: the durable "a human plan review is still owed" obligation
|
|
341
|
+
# for a spec_checkpoint story. Latched at the story's first (leg-1) dispatch —
|
|
342
|
+
# BEFORE the session runs and keyed off the entry's spec_checkpoint flag, not
|
|
343
|
+
# the leg's on-disk status or result — so it survives a crash, a non-fixable
|
|
344
|
+
# retry, or a skill that overran `Halt after planning.`, none of which the
|
|
345
|
+
# on-disk-status-keyed _plan_halt_leg / result-keyed plan_checkpoint_pending
|
|
346
|
+
# carry across. Cleared ONLY when a plan-review pause actually raises (the
|
|
347
|
+
# obligation is discharged). While set after a dev leg that did not itself pause,
|
|
348
|
+
# StoriesEngine pauses before commit so the story can never commit un-reviewed.
|
|
349
|
+
plan_review_owed: bool = False
|
|
350
|
+
# stories mode only: the fixed slug ("unresolved" / "ambiguous") of a pre-planning
|
|
351
|
+
# halt sentinel this task was detected as — recorded at detection time (pick-time
|
|
352
|
+
# wedge or post-dev read-back), NOT re-derived from the spec_file basename at
|
|
353
|
+
# re-arm. runs.rearm_escalation deletes a sentinel only when this is set, so a real
|
|
354
|
+
# story spec that merely happens to be named `<key>-unresolved.md`, or a
|
|
355
|
+
# non-sentinel escalation whose spec matches the convention, is status-flipped and
|
|
356
|
+
# kept, never deleted. "" = not a sentinel. Survives the round-trip.
|
|
357
|
+
sentinel_kind: str = ""
|
|
358
|
+
# intent-gap patch-restore re-drive (Froid Plane #2564): a repo-relative-or-
|
|
359
|
+
# absolute path to the patch file froid-build-auto saved of the reverted attempt.
|
|
360
|
+
# Latched by runs.rearm_escalation when the human confirms the attempted reading
|
|
361
|
+
# was correct; the engine re-applies it onto the baseline after every reset of
|
|
362
|
+
# the re-drive so the re-driven session resumes review (step-04) on the restored
|
|
363
|
+
# diff, and clears it once the corrected work commits. None = ordinary
|
|
364
|
+
# from-scratch re-drive. Survives the resume serialization round-trip.
|
|
365
|
+
restore_patch: str | None = None
|
|
366
|
+
# sweep bundles only: the deferred-work ids this task closes and the
|
|
367
|
+
# rendered intent file handed to dev sessions
|
|
368
|
+
dw_ids: list[str] = field(default_factory=list)
|
|
369
|
+
bundle_file: str | None = None
|
|
370
|
+
# worktree-isolation mode only (scm.isolation = "worktree"): the unit's
|
|
371
|
+
# mounted worktree dir and branch, recorded so a paused/crashed run can
|
|
372
|
+
# reconstruct or discard the in-flight worktree on resume.
|
|
373
|
+
worktree_path: str = ""
|
|
374
|
+
branch: str = ""
|
|
375
|
+
sessions: list[SessionRecord] = field(default_factory=list)
|
|
376
|
+
tokens: TokenUsage = field(default_factory=TokenUsage)
|
|
377
|
+
# latched the first time this story's cost-weighted spend crossed
|
|
378
|
+
# limits.max_tokens_per_story at a session boundary, so the advisory notice
|
|
379
|
+
# fires once per STORY rather than once per session after the crossing
|
|
380
|
+
# (every later session of an overrunning story is over the cap too). Never
|
|
381
|
+
# cleared: the crossing is a fact about the story's spend, and the raw
|
|
382
|
+
# counts it was computed from stay in `tokens`. Persisted precisely so a
|
|
383
|
+
# resumed run does not re-notify what the pre-pause process already did.
|
|
384
|
+
token_budget_warned: bool = False
|
|
385
|
+
|
|
386
|
+
@property
|
|
387
|
+
def terminal(self) -> bool:
|
|
388
|
+
return self.phase in TERMINAL_PHASES
|
|
389
|
+
|
|
390
|
+
def record_session(self, record: SessionRecord) -> None:
|
|
391
|
+
self.sessions.append(record)
|
|
392
|
+
if record.usage:
|
|
393
|
+
self.tokens.add(record.usage)
|
|
394
|
+
|
|
395
|
+
def attach_session_usage(self, task_id: str, usage: TokenUsage | None) -> None:
|
|
396
|
+
"""Fold usage into the most recent session for `task_id`. Usage is
|
|
397
|
+
best-effort metadata attached after the session itself is saved, so a
|
|
398
|
+
failed usage read never costs the recorded session."""
|
|
399
|
+
if usage is None:
|
|
400
|
+
return
|
|
401
|
+
for record in reversed(self.sessions):
|
|
402
|
+
if record.task_id != task_id:
|
|
403
|
+
continue
|
|
404
|
+
if record.usage is None:
|
|
405
|
+
record.usage = usage
|
|
406
|
+
self.tokens.add(usage)
|
|
407
|
+
return
|
|
408
|
+
raise KeyError(task_id)
|
|
409
|
+
|
|
410
|
+
def to_dict(self) -> dict[str, Any]:
|
|
411
|
+
return {
|
|
412
|
+
"story_key": self.story_key,
|
|
413
|
+
"epic": self.epic,
|
|
414
|
+
"phase": str(self.phase),
|
|
415
|
+
"attempt": self.attempt,
|
|
416
|
+
"review_cycle": self.review_cycle,
|
|
417
|
+
"followup_reviews_spent": self.followup_reviews_spent,
|
|
418
|
+
"generation": self.generation,
|
|
419
|
+
"followup_review_recommended": self.followup_review_recommended,
|
|
420
|
+
"baseline_commit": self.baseline_commit,
|
|
421
|
+
"baseline_untracked": self.baseline_untracked,
|
|
422
|
+
"baseline_ledger_digest": self.baseline_ledger_digest,
|
|
423
|
+
"pre_harvest_ledger": self.pre_harvest_ledger,
|
|
424
|
+
"pre_harvest_ledger_captured": self.pre_harvest_ledger_captured,
|
|
425
|
+
"post_engine_ledger_digest": self.post_engine_ledger_digest,
|
|
426
|
+
"harvest_wrote_ledger": self.harvest_wrote_ledger,
|
|
427
|
+
"ledger_changed_before_harvest": self.ledger_changed_before_harvest,
|
|
428
|
+
"harvested_deferrals": self.harvested_deferrals,
|
|
429
|
+
"bundle_closes_intended": self.bundle_closes_intended,
|
|
430
|
+
"refiled_followups": self.refiled_followups,
|
|
431
|
+
"story_closes_intended": self.story_closes_intended,
|
|
432
|
+
"board_advance_intended": self.board_advance_intended,
|
|
433
|
+
"accepted_dev_session_index": self.accepted_dev_session_index,
|
|
434
|
+
"harvest_carry_commit_pending": self.harvest_carry_commit_pending,
|
|
435
|
+
"isolated_ledger_carried": self.isolated_ledger_carried,
|
|
436
|
+
"spec_file": self._serialized_worktree_path(self.spec_file),
|
|
437
|
+
"dispatched_spec_file": self._serialized_worktree_path(self.dispatched_spec_file),
|
|
438
|
+
"dispatched_spec_snapshot": (
|
|
439
|
+
base64.b64encode(self.dispatched_spec_snapshot).decode("ascii")
|
|
440
|
+
if self.dispatched_spec_snapshot is not None
|
|
441
|
+
else None
|
|
442
|
+
),
|
|
443
|
+
"commit_sha": self.commit_sha,
|
|
444
|
+
"operator_actions": self.operator_actions,
|
|
445
|
+
"defer_reason": self.defer_reason,
|
|
446
|
+
"preserve_ref": self.preserve_ref,
|
|
447
|
+
"preserve_partial": self.preserve_partial,
|
|
448
|
+
"rearmed": self.rearmed,
|
|
449
|
+
"resolved_redrive": self.resolved_redrive,
|
|
450
|
+
"plan_checkpoint_pending": self.plan_checkpoint_pending,
|
|
451
|
+
"plan_review_owed": self.plan_review_owed,
|
|
452
|
+
"sentinel_kind": self.sentinel_kind,
|
|
453
|
+
"restore_patch": self.restore_patch,
|
|
454
|
+
"dw_ids": self.dw_ids,
|
|
455
|
+
"bundle_file": self.bundle_file,
|
|
456
|
+
"worktree_path": self.worktree_path,
|
|
457
|
+
"branch": self.branch,
|
|
458
|
+
"sessions": [s.to_dict() for s in self.sessions],
|
|
459
|
+
"tokens": self.tokens.to_dict(),
|
|
460
|
+
"token_budget_warned": self.token_budget_warned,
|
|
461
|
+
}
|
|
462
|
+
|
|
463
|
+
def _serialized_worktree_path(self, path: str | None) -> str | None:
|
|
464
|
+
"""Persist a worktree-local spec path relative to its mounted root.
|
|
465
|
+
|
|
466
|
+
Both the accepted/result spec and the attempt-owned dispatched spec use
|
|
467
|
+
this one normalization path so their state.json representations cannot
|
|
468
|
+
drift. In-place and outside-worktree paths remain verbatim.
|
|
469
|
+
"""
|
|
470
|
+
if not path or not self.worktree_path:
|
|
471
|
+
return path
|
|
472
|
+
try:
|
|
473
|
+
# as_posix: persist the relative path with forward slashes so state.json
|
|
474
|
+
# stays portable across OSes (matches the in-worktree spec layout).
|
|
475
|
+
return Path(path).relative_to(self.worktree_path).as_posix()
|
|
476
|
+
except ValueError:
|
|
477
|
+
return path # spec lives outside the worktree; keep absolute
|
|
478
|
+
|
|
479
|
+
def release_spec_paths_from_mount(self) -> None:
|
|
480
|
+
"""Give up the spec ownership a mount being DISCARDED carried.
|
|
481
|
+
|
|
482
|
+
The counterpart to :meth:`rebase_spec_paths_on`, and deliberately not its
|
|
483
|
+
exact inverse — the two fields part company here because their roles do:
|
|
484
|
+
|
|
485
|
+
* `dispatched_spec_file` / `dispatched_spec_snapshot` are the ATTEMPT's
|
|
486
|
+
binding, the pair `recovery_flow` restores bytes through. The attempt died
|
|
487
|
+
with its tree, so the binding has nothing left to name; clearing both
|
|
488
|
+
together keeps the authority pair whole (a path without its snapshot is the
|
|
489
|
+
one shape `_bind_dispatched_spec_for_attempt` never persists).
|
|
490
|
+
* `spec_file` is the ACCEPTED artifact and outlives the attempt. The
|
|
491
|
+
replacement mount will carry the same story's spec at the same
|
|
492
|
+
mount-relative place, so the relative spelling is the one that re-resolves
|
|
493
|
+
onto it — `verify.resolve_spec_path` probes a relative value against the
|
|
494
|
+
live workspace and passes an absolute one through untouched.
|
|
495
|
+
|
|
496
|
+
Leaving `spec_file` absolute into the deleted mount is what made the fresh
|
|
497
|
+
attempt start UNBOUND: `_dispatched_spec_for_attempt` resolves it
|
|
498
|
+
`strict=True`, the dead path raises, and the miss is silent because an
|
|
499
|
+
unbound attempt is a legal state. `_record_dev_spec` cannot repair it either
|
|
500
|
+
— it no-ops while `spec_file` is set.
|
|
501
|
+
|
|
502
|
+
Uses the same relativization as `to_dict`, so the discarded-mount spelling
|
|
503
|
+
and the persisted one cannot drift, which also means a spec OUTSIDE the mount
|
|
504
|
+
stays verbatim: it was never the mount's to give up. MUST be called while
|
|
505
|
+
`worktree_path` still names the mount.
|
|
506
|
+
"""
|
|
507
|
+
self.dispatched_spec_file = None
|
|
508
|
+
self.dispatched_spec_snapshot = None
|
|
509
|
+
self.spec_file = self._serialized_worktree_path(self.spec_file)
|
|
510
|
+
|
|
511
|
+
def release_mount_owned_state(self) -> None:
|
|
512
|
+
"""Give up EVERYTHING a mount owned: its spec ownership and the measurements
|
|
513
|
+
taken inside it.
|
|
514
|
+
|
|
515
|
+
One method because the two callers that stop using a mount — the restart
|
|
516
|
+
discard and the isolation-flip arm — must give up the same set, and the second
|
|
517
|
+
was written releasing only the spec half. That half-release is not a smaller
|
|
518
|
+
version of the same thing, it is a different bug: `baseline_commit` and
|
|
519
|
+
`baseline_untracked` are stamped from `self.workspace.root` (the unit under
|
|
520
|
+
isolation), so leaving them set hands unit-mount operands to
|
|
521
|
+
`recovery_flow.rollback_or_pause` running against the MAIN checkout. Neither
|
|
522
|
+
fails loud there — linked worktrees share the object database, so the baseline
|
|
523
|
+
still resolves and a reset onto it succeeds, while a fresh worktree is a
|
|
524
|
+
tracked-only checkout whose empty untracked snapshot makes
|
|
525
|
+
`verify._rollback_cleanup_plan` compute `untracked_files(repo) -
|
|
526
|
+
baseline_untracked` as every untracked file in the operator's own checkout.
|
|
527
|
+
Under an auto-recovering cause those are DELETED.
|
|
528
|
+
|
|
529
|
+
Costs the re-run nothing: `_dev_phase` re-stamps both from whatever workspace
|
|
530
|
+
it re-enters with, so clearing turns the `baseline_commit` leg into a correct
|
|
531
|
+
no-op instead of a probe of the wrong tree.
|
|
532
|
+
|
|
533
|
+
MUST be called while `worktree_path` still names the mount — the spec
|
|
534
|
+
relativization is measured against it.
|
|
535
|
+
"""
|
|
536
|
+
self.release_spec_paths_from_mount()
|
|
537
|
+
self.baseline_commit = None
|
|
538
|
+
self.baseline_untracked = None
|
|
539
|
+
|
|
540
|
+
def rebase_spec_paths_on(self, root: Path) -> None:
|
|
541
|
+
"""Re-absolutize both spec-ownership paths against the tree that owns them.
|
|
542
|
+
|
|
543
|
+
The read-side inverse of :meth:`_serialized_worktree_path`, and the single
|
|
544
|
+
implementation of that rule: `to_dict` persists a worktree-local spec
|
|
545
|
+
RELATIVE to the mount and `from_dict` reads it back raw, so a consumer that
|
|
546
|
+
resolves the raw value against anything else names the wrong tree. The main
|
|
547
|
+
checkout carries the same `_froid-output/...` layout, so that wrong tree
|
|
548
|
+
answers `is_file()` and passes containment — the failure is silent, not an
|
|
549
|
+
error.
|
|
550
|
+
|
|
551
|
+
Both fields move together because they are one asymmetry: `spec_file` is the
|
|
552
|
+
accepted/result artifact and `dispatched_spec_file` the attempt-owned input,
|
|
553
|
+
and a caller re-anchoring one and not the other leaves a task naming two
|
|
554
|
+
trees at once.
|
|
555
|
+
|
|
556
|
+
Idempotent: an absolute value is already anchored (a spec outside the mount
|
|
557
|
+
is persisted verbatim) and passes through untouched, so re-running this
|
|
558
|
+
against the same root cannot double-join. `root` is the tree the values were
|
|
559
|
+
persisted relative to — `task.worktree_path` — never the caller's cwd or
|
|
560
|
+
project.
|
|
561
|
+
"""
|
|
562
|
+
self.spec_file = _rebased_on(self.spec_file, root)
|
|
563
|
+
self.dispatched_spec_file = _rebased_on(self.dispatched_spec_file, root)
|
|
564
|
+
|
|
565
|
+
@classmethod
|
|
566
|
+
def from_dict(cls, d: dict[str, Any]) -> "StoryTask":
|
|
567
|
+
dispatched_spec_snapshot = d.get("dispatched_spec_snapshot")
|
|
568
|
+
if dispatched_spec_snapshot is not None:
|
|
569
|
+
try:
|
|
570
|
+
dispatched_spec_snapshot = base64.b64decode(
|
|
571
|
+
str(dispatched_spec_snapshot).encode("ascii"),
|
|
572
|
+
validate=True,
|
|
573
|
+
)
|
|
574
|
+
except ValueError as exc:
|
|
575
|
+
raise ValueError(
|
|
576
|
+
f"story {d.get('story_key')!r}: dispatched_spec_snapshot " "is not valid base64"
|
|
577
|
+
) from exc
|
|
578
|
+
return cls(
|
|
579
|
+
story_key=d["story_key"],
|
|
580
|
+
epic=int(d["epic"]),
|
|
581
|
+
phase=Phase(d["phase"]),
|
|
582
|
+
attempt=int(d.get("attempt", 0)),
|
|
583
|
+
review_cycle=int(d.get("review_cycle", 0)),
|
|
584
|
+
followup_reviews_spent=int(d.get("followup_reviews_spent", 0)),
|
|
585
|
+
generation=int(d.get("generation", 0)),
|
|
586
|
+
followup_review_recommended=bool(d.get("followup_review_recommended", False)),
|
|
587
|
+
baseline_commit=d.get("baseline_commit"),
|
|
588
|
+
baseline_untracked=(
|
|
589
|
+
[str(p) for p in d["baseline_untracked"]]
|
|
590
|
+
if d.get("baseline_untracked") is not None
|
|
591
|
+
else None
|
|
592
|
+
),
|
|
593
|
+
baseline_ledger_digest=(
|
|
594
|
+
str(d.get("baseline_ledger_digest"))
|
|
595
|
+
if d.get("baseline_ledger_digest") is not None
|
|
596
|
+
else None
|
|
597
|
+
),
|
|
598
|
+
pre_harvest_ledger=(
|
|
599
|
+
str(d.get("pre_harvest_ledger"))
|
|
600
|
+
if d.get("pre_harvest_ledger") is not None
|
|
601
|
+
else None
|
|
602
|
+
),
|
|
603
|
+
pre_harvest_ledger_captured=bool(d.get("pre_harvest_ledger_captured", False)),
|
|
604
|
+
post_engine_ledger_digest=(
|
|
605
|
+
str(d.get("post_engine_ledger_digest"))
|
|
606
|
+
if d.get("post_engine_ledger_digest") is not None
|
|
607
|
+
else None
|
|
608
|
+
),
|
|
609
|
+
harvest_wrote_ledger=bool(d.get("harvest_wrote_ledger", False)),
|
|
610
|
+
ledger_changed_before_harvest=bool(d.get("ledger_changed_before_harvest", False)),
|
|
611
|
+
harvested_deferrals=[deepcopy(dict(item)) for item in d.get("harvested_deferrals", [])],
|
|
612
|
+
bundle_closes_intended=[str(i) for i in d.get("bundle_closes_intended", [])],
|
|
613
|
+
refiled_followups=[deepcopy(dict(item)) for item in d.get("refiled_followups", [])],
|
|
614
|
+
story_closes_intended=[str(i) for i in d.get("story_closes_intended", [])],
|
|
615
|
+
board_advance_intended=(
|
|
616
|
+
str(d["board_advance_intended"])
|
|
617
|
+
if d.get("board_advance_intended") is not None
|
|
618
|
+
else None
|
|
619
|
+
),
|
|
620
|
+
accepted_dev_session_index=(
|
|
621
|
+
int(d["accepted_dev_session_index"])
|
|
622
|
+
if d.get("accepted_dev_session_index") is not None
|
|
623
|
+
else None
|
|
624
|
+
),
|
|
625
|
+
harvest_carry_commit_pending=bool(d.get("harvest_carry_commit_pending", False)),
|
|
626
|
+
isolated_ledger_carried=bool(d.get("isolated_ledger_carried", False)),
|
|
627
|
+
spec_file=d.get("spec_file"),
|
|
628
|
+
dispatched_spec_file=d.get("dispatched_spec_file"),
|
|
629
|
+
dispatched_spec_snapshot=dispatched_spec_snapshot,
|
|
630
|
+
commit_sha=d.get("commit_sha"),
|
|
631
|
+
operator_actions=[str(a) for a in d.get("operator_actions", [])],
|
|
632
|
+
defer_reason=d.get("defer_reason"),
|
|
633
|
+
preserve_ref=d.get("preserve_ref"),
|
|
634
|
+
preserve_partial=bool(d.get("preserve_partial", False)),
|
|
635
|
+
rearmed=bool(d.get("rearmed", False)),
|
|
636
|
+
resolved_redrive=bool(d.get("resolved_redrive", False)),
|
|
637
|
+
plan_checkpoint_pending=bool(d.get("plan_checkpoint_pending", False)),
|
|
638
|
+
plan_review_owed=bool(d.get("plan_review_owed", False)),
|
|
639
|
+
sentinel_kind=str(d.get("sentinel_kind", "")),
|
|
640
|
+
restore_patch=d.get("restore_patch"),
|
|
641
|
+
dw_ids=[str(i) for i in d.get("dw_ids", [])],
|
|
642
|
+
bundle_file=d.get("bundle_file"),
|
|
643
|
+
worktree_path=str(d.get("worktree_path", "")),
|
|
644
|
+
branch=str(d.get("branch", "")),
|
|
645
|
+
sessions=[SessionRecord.from_dict(s) for s in d.get("sessions", [])],
|
|
646
|
+
tokens=TokenUsage.from_dict(d.get("tokens", {})),
|
|
647
|
+
token_budget_warned=bool(d.get("token_budget_warned", False)),
|
|
648
|
+
)
|
|
649
|
+
|
|
650
|
+
|
|
651
|
+
@dataclass
|
|
652
|
+
class RunState:
|
|
653
|
+
run_id: str
|
|
654
|
+
project: str
|
|
655
|
+
started_at: str
|
|
656
|
+
# The git root this run's code work happens in — `paths.repo_root`, which is
|
|
657
|
+
# `paths.project` unless `_froid/bmm/config.yaml` sets a `repo_root:` override.
|
|
658
|
+
# Persisted because `runs.rearm_escalation` runs OUT OF PROCESS from the engine
|
|
659
|
+
# and had only `project` to reach for, so it advanced the attempt baseline by
|
|
660
|
+
# reading HEAD of a repo the proof-of-work gate never measures. Empty means a
|
|
661
|
+
# state.json written before this field existed; `code_root` then falls back to
|
|
662
|
+
# `project`, which is exactly the pre-upgrade behavior and the correct answer
|
|
663
|
+
# for every run without the override.
|
|
664
|
+
repo_root: str = ""
|
|
665
|
+
policy_snapshot: dict[str, Any] = field(default_factory=dict)
|
|
666
|
+
# SECONDARY copy of the host-exec baseline (#498) — runsetup.config_digest over
|
|
667
|
+
# the agent-writable config that reaches HOST code execution: verify commands,
|
|
668
|
+
# the resolved launch binary/args/env, the plugin allowlist (#461 point 4).
|
|
669
|
+
#
|
|
670
|
+
# The one resume TRUSTS is out of the tree (`runs.write_trusted_config_digest`),
|
|
671
|
+
# because a baseline whose whole job is to police the agent-writable tree cannot
|
|
672
|
+
# live in it — a session that rewrote policy.toml could blank this field in the
|
|
673
|
+
# same breath and silence the warning `resume` owes the operator. So this copy is
|
|
674
|
+
# never preferred: `_resume_paused_run` consults it ONLY when the state root
|
|
675
|
+
# holds no file for the run, which is what keeps rewriting it pointless (the
|
|
676
|
+
# #498 attack test asserts exactly that).
|
|
677
|
+
#
|
|
678
|
+
# It is still written, and must be, for the two cases where the out-of-tree file
|
|
679
|
+
# is honestly absent rather than tampered away — in both, this copy is the run's
|
|
680
|
+
# only surviving pin:
|
|
681
|
+
# * the state root is keyed by the project's RESOLVED PATH (`runs.project_tag`),
|
|
682
|
+
# so moving or renaming the project keys the run somewhere new and orphans
|
|
683
|
+
# its state subtree (FEATURES.md documents the GC half of this). state.json
|
|
684
|
+
# lives in the run dir and travels with it. Same for a FROID_LOOP_STATE_DIR
|
|
685
|
+
# that changes between launch and resume.
|
|
686
|
+
# * a run PAUSED before #498 has its baseline here and nowhere else; the first
|
|
687
|
+
# resume under this code reads it and mints the out-of-tree file.
|
|
688
|
+
# Dropping it would turn "this run has a pin" into "this run has none" in all of
|
|
689
|
+
# them. Empty means what it always did — no prior pin, hence no warning — which
|
|
690
|
+
# is why the resume compare is guarded on non-emptiness. The auto-sweep gate has
|
|
691
|
+
# never read this field: it compares against its own in-memory closure baseline,
|
|
692
|
+
# which no session can reach.
|
|
693
|
+
trusted_config_digest: str = ""
|
|
694
|
+
current_epic: int | None = None
|
|
695
|
+
# the run's story scope + cap, as passed on the launching CLI (`--epic`,
|
|
696
|
+
# `--story`, `--max-stories`). Persisted so `resume` rebuilds the Engine with
|
|
697
|
+
# the SAME selector — otherwise a resumed `--epic N` run silently widens to
|
|
698
|
+
# every epic and can jump out of its scope at the next pick.
|
|
699
|
+
epic_filter: int | None = None
|
|
700
|
+
story_filter: str | None = None
|
|
701
|
+
max_stories: int | None = None
|
|
702
|
+
paused_reason: str | None = None
|
|
703
|
+
paused_stage: str | None = None
|
|
704
|
+
paused_story_key: str | None = None
|
|
705
|
+
finished: bool = False
|
|
706
|
+
# deliberately stopped (froid-loop stop / engine SIGTERM); distinct from a
|
|
707
|
+
# crash. Resume clears it via clear_pause(), so a stopped run is resumable.
|
|
708
|
+
stopped: bool = False
|
|
709
|
+
# an unexpected exception escaped Engine.run() and was recorded (crash.txt +
|
|
710
|
+
# run-crash journal). Distinct from `stopped`; resume clears it via
|
|
711
|
+
# clear_pause() so a crashed run re-arms like a stopped one. crash_error is a
|
|
712
|
+
# short "Type: message" for display; the full traceback lives in crash.txt.
|
|
713
|
+
crashed: bool = False
|
|
714
|
+
crash_error: str | None = None
|
|
715
|
+
run_type: str = "story" # "story" | "sweep" — resume/status dispatch on it
|
|
716
|
+
# story-queue source (policy.StoriesPolicy.source), pinned at run start so
|
|
717
|
+
# resume/resolve rebuild the right engine (StoriesEngine vs the sprint Engine)
|
|
718
|
+
# without re-reading policy — a policy edit mid-run must not switch a live run's
|
|
719
|
+
# mode. `run_type` stays "story" for both; `source` selects the picker.
|
|
720
|
+
source: str = "sprint-status"
|
|
721
|
+
# stories mode only: the project-relative (or absolute) spec folder holding
|
|
722
|
+
# stories.yaml + SPEC.md. Empty under sprint-status.
|
|
723
|
+
spec_folder: str = ""
|
|
724
|
+
# sweep runs only: the triage->bundles cycle in progress; 1 maps to the
|
|
725
|
+
# legacy (unsuffixed) artifact names so old paused runs resume unchanged
|
|
726
|
+
sweep_cycle: int = 1
|
|
727
|
+
# auto-sweep triggers already fired this run (e.g. "epic-1", "run-end");
|
|
728
|
+
# guards re-fire on resume
|
|
729
|
+
sweeps_triggered: list[str] = field(default_factory=list)
|
|
730
|
+
# auto-sweep triggers this run did NOT deliver, trigger -> SWEEP_REFUSED_*.
|
|
731
|
+
# Kept apart from sweeps_triggered rather than folded into it: that list is
|
|
732
|
+
# the re-fire latch, and widening it to a mapping would silently degrade the
|
|
733
|
+
# per-element sanitizer loop in diagnostics.py. A trigger may appear in both
|
|
734
|
+
# (SWEEP_REFUSED_FAILED = a child that started and then failed).
|
|
735
|
+
sweeps_refused: dict[str, str] = field(default_factory=dict)
|
|
736
|
+
# worktree-isolation mode only: the branch every unit merges back into,
|
|
737
|
+
# resolved once at run start (default = the branch checked out then) and
|
|
738
|
+
# pinned so resume keeps targeting the same branch.
|
|
739
|
+
target_branch: str = ""
|
|
740
|
+
# free-form scratch space shared across plugin hooks (HookContext.shared).
|
|
741
|
+
# Persisted so a plugin's cross-stage state survives pause/resume; values
|
|
742
|
+
# MUST be JSON-serializable. Empty + untouched on a zero-plugin run.
|
|
743
|
+
plugin_shared: dict[str, Any] = field(default_factory=dict)
|
|
744
|
+
tasks: dict[str, StoryTask] = field(default_factory=dict)
|
|
745
|
+
|
|
746
|
+
@property
|
|
747
|
+
def paused(self) -> bool:
|
|
748
|
+
return self.paused_reason is not None
|
|
749
|
+
|
|
750
|
+
@property
|
|
751
|
+
def code_root(self) -> Path:
|
|
752
|
+
"""The tree git runs against for this run — ``repo_root`` when the run
|
|
753
|
+
recorded one, else ``project``.
|
|
754
|
+
|
|
755
|
+
The single reader of the pair, so an out-of-process consumer
|
|
756
|
+
(``runs.rearm_escalation``) cannot pick the wrong one, and a pre-upgrade
|
|
757
|
+
state.json (empty ``repo_root``) degrades to precisely what it did before
|
|
758
|
+
rather than to a path that does not exist."""
|
|
759
|
+
return Path(self.repo_root or self.project)
|
|
760
|
+
|
|
761
|
+
def handled_keys(self) -> set[str]:
|
|
762
|
+
"""Story keys this run already drove to a terminal phase."""
|
|
763
|
+
return {k for k, t in self.tasks.items() if t.terminal}
|
|
764
|
+
|
|
765
|
+
def clear_pause(self) -> None:
|
|
766
|
+
self.paused_reason = None
|
|
767
|
+
self.paused_stage = None
|
|
768
|
+
self.paused_story_key = None
|
|
769
|
+
self.stopped = False
|
|
770
|
+
self.crashed = False
|
|
771
|
+
self.crash_error = None
|
|
772
|
+
|
|
773
|
+
def cache_read_weight(self) -> float:
|
|
774
|
+
"""The run's cache-read weight from its persisted policy snapshot; the
|
|
775
|
+
product default (policy.LimitsPolicy.cache_read_weight = 0.1) when the
|
|
776
|
+
snapshot predates the field or is malformed. Lets the TUI show the same
|
|
777
|
+
weighted total the engine's budget uses without importing Policy.
|
|
778
|
+
|
|
779
|
+
The snapshot is re-stamped at every engine start (run, sweep, resume), so
|
|
780
|
+
on a resumed run this is the *resuming* process's weight, matching what
|
|
781
|
+
that process enforces. Edit the weight and resume and the run's whole
|
|
782
|
+
accumulated history re-weights — totals are recomputed from raw counts,
|
|
783
|
+
and the budget has always judged cumulative counts at the live weight."""
|
|
784
|
+
limits = self.policy_snapshot.get("limits")
|
|
785
|
+
if isinstance(limits, dict):
|
|
786
|
+
try:
|
|
787
|
+
return float(limits["cache_read_weight"])
|
|
788
|
+
except (KeyError, TypeError, ValueError):
|
|
789
|
+
pass
|
|
790
|
+
return 0.1
|
|
791
|
+
|
|
792
|
+
def to_dict(self) -> dict[str, Any]:
|
|
793
|
+
return {
|
|
794
|
+
"run_id": self.run_id,
|
|
795
|
+
"project": self.project,
|
|
796
|
+
"repo_root": self.repo_root,
|
|
797
|
+
"started_at": self.started_at,
|
|
798
|
+
"policy_snapshot": self.policy_snapshot,
|
|
799
|
+
"trusted_config_digest": self.trusted_config_digest,
|
|
800
|
+
"current_epic": self.current_epic,
|
|
801
|
+
"epic_filter": self.epic_filter,
|
|
802
|
+
"story_filter": self.story_filter,
|
|
803
|
+
"max_stories": self.max_stories,
|
|
804
|
+
"paused_reason": self.paused_reason,
|
|
805
|
+
"paused_stage": self.paused_stage,
|
|
806
|
+
"paused_story_key": self.paused_story_key,
|
|
807
|
+
"finished": self.finished,
|
|
808
|
+
"stopped": self.stopped,
|
|
809
|
+
"crashed": self.crashed,
|
|
810
|
+
"crash_error": self.crash_error,
|
|
811
|
+
"run_type": self.run_type,
|
|
812
|
+
"source": self.source,
|
|
813
|
+
"spec_folder": self.spec_folder,
|
|
814
|
+
"sweep_cycle": self.sweep_cycle,
|
|
815
|
+
"sweeps_triggered": self.sweeps_triggered,
|
|
816
|
+
"sweeps_refused": self.sweeps_refused,
|
|
817
|
+
"target_branch": self.target_branch,
|
|
818
|
+
"plugin_shared": self.plugin_shared,
|
|
819
|
+
"tasks": {k: t.to_dict() for k, t in self.tasks.items()},
|
|
820
|
+
}
|
|
821
|
+
|
|
822
|
+
@classmethod
|
|
823
|
+
def from_dict(cls, d: dict[str, Any]) -> "RunState":
|
|
824
|
+
return cls(
|
|
825
|
+
run_id=d["run_id"],
|
|
826
|
+
project=d["project"],
|
|
827
|
+
repo_root=str(d.get("repo_root", "")),
|
|
828
|
+
started_at=d["started_at"],
|
|
829
|
+
policy_snapshot=d.get("policy_snapshot", {}),
|
|
830
|
+
trusted_config_digest=str(d.get("trusted_config_digest", "")),
|
|
831
|
+
current_epic=d.get("current_epic"),
|
|
832
|
+
epic_filter=d.get("epic_filter"),
|
|
833
|
+
story_filter=d.get("story_filter"),
|
|
834
|
+
max_stories=d.get("max_stories"),
|
|
835
|
+
paused_reason=d.get("paused_reason"),
|
|
836
|
+
paused_stage=d.get("paused_stage"),
|
|
837
|
+
paused_story_key=d.get("paused_story_key"),
|
|
838
|
+
finished=bool(d.get("finished", False)),
|
|
839
|
+
stopped=bool(d.get("stopped", False)),
|
|
840
|
+
crashed=bool(d.get("crashed", False)),
|
|
841
|
+
crash_error=d.get("crash_error"),
|
|
842
|
+
run_type=str(d.get("run_type", "story")),
|
|
843
|
+
source=str(d.get("source", "sprint-status")),
|
|
844
|
+
spec_folder=str(d.get("spec_folder", "")),
|
|
845
|
+
sweep_cycle=int(d.get("sweep_cycle", 1)),
|
|
846
|
+
sweeps_triggered=[str(s) for s in d.get("sweeps_triggered", [])],
|
|
847
|
+
sweeps_refused={str(k): str(v) for k, v in d.get("sweeps_refused", {}).items()},
|
|
848
|
+
target_branch=str(d.get("target_branch", "")),
|
|
849
|
+
plugin_shared=dict(d.get("plugin_shared", {})),
|
|
850
|
+
tasks={k: StoryTask.from_dict(t) for k, t in d.get("tasks", {}).items()},
|
|
851
|
+
)
|
|
852
|
+
|
|
853
|
+
|
|
854
|
+
@dataclass(frozen=True)
|
|
855
|
+
class VerifyOutcome:
|
|
856
|
+
ok: bool
|
|
857
|
+
reason: str = ""
|
|
858
|
+
severity: str = "" # "" | "CRITICAL" | "PREFERENCE" — set when not retryable
|
|
859
|
+
# fixable failures carry concrete evidence (failing command output) that a
|
|
860
|
+
# feedback-driven repair session can act on; non-fixable retries start over
|
|
861
|
+
fixable: bool = False
|
|
862
|
+
# the failure is the run environment's, not the story's (verify command
|
|
863
|
+
# not found / not executable): no repair session can fix it and every
|
|
864
|
+
# story shares the same commands, so it must never charge attempt budgets
|
|
865
|
+
env_fault: bool = False
|
|
866
|
+
# a session deliberately contradicted a state the orchestrator had already
|
|
867
|
+
# established (a review revoking the sprint sign-off it advanced at dev
|
|
868
|
+
# time): no further session can reconcile it, so it routes to a pause with
|
|
869
|
+
# both sides named rather than to another cycle (#334)
|
|
870
|
+
contradiction: bool = False
|
|
871
|
+
|
|
872
|
+
@classmethod
|
|
873
|
+
def passed(cls) -> "VerifyOutcome":
|
|
874
|
+
return cls(ok=True)
|
|
875
|
+
|
|
876
|
+
@classmethod
|
|
877
|
+
def retry(cls, reason: str, fixable: bool = False) -> "VerifyOutcome":
|
|
878
|
+
return cls(ok=False, reason=reason, fixable=fixable)
|
|
879
|
+
|
|
880
|
+
@classmethod
|
|
881
|
+
def escalate(
|
|
882
|
+
cls,
|
|
883
|
+
reason: str,
|
|
884
|
+
severity: str = "CRITICAL",
|
|
885
|
+
env_fault: bool = False,
|
|
886
|
+
contradiction: bool = False,
|
|
887
|
+
) -> "VerifyOutcome":
|
|
888
|
+
return cls(
|
|
889
|
+
ok=False,
|
|
890
|
+
reason=reason,
|
|
891
|
+
severity=severity,
|
|
892
|
+
env_fault=env_fault,
|
|
893
|
+
contradiction=contradiction,
|
|
894
|
+
)
|
|
895
|
+
|
|
896
|
+
@property
|
|
897
|
+
def retryable(self) -> bool:
|
|
898
|
+
return not self.ok and not self.severity
|