syncade 0.6.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- syncade/__init__.py +3 -0
- syncade/__main__.py +6 -0
- syncade/adapters/__init__.py +0 -0
- syncade/adapters/anthropic.py +457 -0
- syncade/adapters/base.py +221 -0
- syncade/adapters/fake.py +73 -0
- syncade/adapters/fake_common.py +29 -0
- syncade/adapters/fake_producer_audit_draft.py +460 -0
- syncade/adapters/fake_reviewer_synth.py +310 -0
- syncade/adapters/openai.py +484 -0
- syncade/adapters/openai_parsing.py +119 -0
- syncade/adapters/producer.py +221 -0
- syncade/adapters/producer_anthropic.py +300 -0
- syncade/adapters/producer_openai.py +226 -0
- syncade/adapters/registry.py +81 -0
- syncade/auth_check.py +554 -0
- syncade/auth_preflight.py +342 -0
- syncade/base_resolution.py +214 -0
- syncade/billing.py +141 -0
- syncade/checks_config.py +113 -0
- syncade/cli/__init__.py +546 -0
- syncade/cli/auth_gate.py +59 -0
- syncade/cli/config_keys.py +135 -0
- syncade/cli/config_list.py +82 -0
- syncade/cli/config_menu_rows.py +166 -0
- syncade/cli/config_mode.py +609 -0
- syncade/cli/config_overrides.py +122 -0
- syncade/cli/config_tui.py +476 -0
- syncade/cli/doctor_mode.py +72 -0
- syncade/cli/gc_mode.py +109 -0
- syncade/cli/install_skill.py +514 -0
- syncade/cli/metrics_mode.py +363 -0
- syncade/cli/modes.py +573 -0
- syncade/cli/parser.py +450 -0
- syncade/cli/parser_types.py +137 -0
- syncade/cli/paths.py +38 -0
- syncade/cli/preflight_paths.py +90 -0
- syncade/cli/resolve.py +116 -0
- syncade/cli/resume_mode.py +324 -0
- syncade/cli/toml_writer.py +410 -0
- syncade/cli/validate.py +421 -0
- syncade/config.py +478 -0
- syncade/config_auth.py +310 -0
- syncade/config_cold.py +209 -0
- syncade/config_gc.py +55 -0
- syncade/config_loader.py +182 -0
- syncade/config_loop.py +282 -0
- syncade/config_producer.py +222 -0
- syncade/config_retry.py +49 -0
- syncade/config_types.py +59 -0
- syncade/diff_filter.py +437 -0
- syncade/dispatcher.py +571 -0
- syncade/doctor.py +425 -0
- syncade/doctor_env.py +218 -0
- syncade/doctor_preview.py +524 -0
- syncade/doctor_types.py +28 -0
- syncade/exit_codes.py +82 -0
- syncade/findings.py +242 -0
- syncade/findings_json.py +456 -0
- syncade/gc.py +211 -0
- syncade/gc_execute.py +372 -0
- syncade/gc_protection.py +129 -0
- syncade/gc_types.py +50 -0
- syncade/gc_worktrees.py +200 -0
- syncade/git_object_id.py +12 -0
- syncade/git_preconditions.py +389 -0
- syncade/logging.py +289 -0
- syncade/metrics/__init__.py +32 -0
- syncade/metrics/aggregate.py +550 -0
- syncade/metrics/schema.py +221 -0
- syncade/orchestrator/__init__.py +61 -0
- syncade/orchestrator/_runs_dir.py +24 -0
- syncade/orchestrator/branch_advance.py +165 -0
- syncade/orchestrator/branch_guard.py +98 -0
- syncade/orchestrator/budget.py +107 -0
- syncade/orchestrator/escalation_coverage.py +81 -0
- syncade/orchestrator/loop.py +611 -0
- syncade/orchestrator/loop_dispatch_check.py +112 -0
- syncade/orchestrator/loop_finalize.py +404 -0
- syncade/orchestrator/loop_preflight.py +131 -0
- syncade/orchestrator/loop_resume.py +91 -0
- syncade/orchestrator/loop_rmtree.py +70 -0
- syncade/orchestrator/loop_round_step.py +599 -0
- syncade/orchestrator/prior_round.py +336 -0
- syncade/orchestrator/producer_phase.py +169 -0
- syncade/orchestrator/results.py +306 -0
- syncade/orchestrator/resume.py +96 -0
- syncade/orchestrator/resume_load.py +483 -0
- syncade/orchestrator/resume_plan.py +554 -0
- syncade/orchestrator/resume_target.py +215 -0
- syncade/orchestrator/resume_types.py +182 -0
- syncade/orchestrator/reviewer_template_failure.py +99 -0
- syncade/orchestrator/round.py +573 -0
- syncade/orchestrator/round_checks.py +91 -0
- syncade/orchestrator/round_no_changes.py +369 -0
- syncade/orchestrator/round_predispatch.py +212 -0
- syncade/orchestrator/verdict.py +279 -0
- syncade/persistence/__init__.py +189 -0
- syncade/persistence/_atomic.py +33 -0
- syncade/persistence/_clusters.py +70 -0
- syncade/persistence/_findings_verdict.py +201 -0
- syncade/persistence/_markdown.py +286 -0
- syncade/persistence/_validation.py +37 -0
- syncade/persistence/checks.py +249 -0
- syncade/persistence/decision_needed.py +289 -0
- syncade/persistence/findings_md.py +389 -0
- syncade/persistence/handoff.py +389 -0
- syncade/persistence/handoff_classify.py +196 -0
- syncade/persistence/last_reviewed.py +67 -0
- syncade/persistence/loop_manifest.py +165 -0
- syncade/persistence/loop_summary.py +352 -0
- syncade/persistence/loop_summary_text.py +428 -0
- syncade/persistence/producer.py +250 -0
- syncade/persistence/reviewer.py +198 -0
- syncade/persistence/round_manifest.py +238 -0
- syncade/persistence/run_init.py +153 -0
- syncade/persistence/run_summary.py +585 -0
- syncade/persistence/run_summary_next_steps.py +443 -0
- syncade/persistence/synth.py +242 -0
- syncade/persistence/test_run.py +152 -0
- syncade/presets.py +36 -0
- syncade/pricing_config.py +72 -0
- syncade/process.py +600 -0
- syncade/producer.py +189 -0
- syncade/producer_attempt.py +463 -0
- syncade/producer_escalation.py +146 -0
- syncade/producer_git.py +199 -0
- syncade/producer_result.py +205 -0
- syncade/prompts.py +448 -0
- syncade/prompts_loader.py +238 -0
- syncade/retry.py +159 -0
- syncade/run_inputs.py +40 -0
- syncade/run_status.py +198 -0
- syncade/selfcheck.py +471 -0
- syncade/skills/claude/README.md +221 -0
- syncade/skills/claude/SKILL.md +625 -0
- syncade/skills/codex/README.md +116 -0
- syncade/skills/codex/SKILL.md +574 -0
- syncade/snapshot.py +598 -0
- syncade/spec_audit.py +437 -0
- syncade/spec_audit_schema.py +190 -0
- syncade/spec_draft.py +423 -0
- syncade/spec_source.py +135 -0
- syncade/synthesis.py +428 -0
- syncade/synthesis_clusters.py +203 -0
- syncade/synthesis_repair.py +230 -0
- syncade/synthesis_schema.py +65 -0
- syncade/synthesizer/__init__.py +38 -0
- syncade/synthesizer/constants.py +33 -0
- syncade/synthesizer/driver.py +531 -0
- syncade/synthesizer/rendering.py +63 -0
- syncade/synthesizer/result.py +73 -0
- syncade/synthesizer/validation.py +421 -0
- syncade/synthesizer/workspace.py +208 -0
- syncade/templates/presets/balanced.toml +13 -0
- syncade/templates/presets/cheap.toml +12 -0
- syncade/templates/presets/thorough.toml +9 -0
- syncade/templates/producer.md +231 -0
- syncade/templates/reviewer.md +279 -0
- syncade/templates/reviewer_adversarial.md +164 -0
- syncade/templates/reviewer_codex.md +165 -0
- syncade/templates/spec_audit.md +168 -0
- syncade/templates/spec_draft.md +62 -0
- syncade/templates/synthesizer.md +204 -0
- syncade/test_runner.py +476 -0
- syncade/test_runner_classify.py +98 -0
- syncade/transcript.py +150 -0
- syncade/usage.py +407 -0
- syncade/worktree.py +497 -0
- syncade/worktree_env.py +133 -0
- syncade/worktree_paths.py +139 -0
- syncade-0.6.2.dist-info/METADATA +314 -0
- syncade-0.6.2.dist-info/RECORD +177 -0
- syncade-0.6.2.dist-info/WHEEL +5 -0
- syncade-0.6.2.dist-info/entry_points.txt +2 -0
- syncade-0.6.2.dist-info/licenses/LICENSE +202 -0
- syncade-0.6.2.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,554 @@
|
|
|
1
|
+
"""Resume planning.
|
|
2
|
+
|
|
3
|
+
Builds the :class:`ResumePlan` for a run directory: classifies per-round phase
|
|
4
|
+
failures, reads the operator decision for an escalated round, derives the
|
|
5
|
+
expected snapshot SHA, and walks the rounds to find the first incomplete one.
|
|
6
|
+
Pure read-from-disk; raises :class:`ResumeError` on malformed/degenerate state.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import json
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
from syncade.exit_codes import FINDINGS_PRESENT, SUCCESS
|
|
15
|
+
from syncade.git_object_id import is_full_git_object_id
|
|
16
|
+
from syncade.persistence import (
|
|
17
|
+
OPERATOR_DECISION_FILENAME,
|
|
18
|
+
RUN_INIT_FILENAME,
|
|
19
|
+
read_operator_decision,
|
|
20
|
+
)
|
|
21
|
+
from syncade.test_runner import is_blocking_check_subprocess_error
|
|
22
|
+
|
|
23
|
+
from .resume_types import (
|
|
24
|
+
LOOP_MANIFEST_FILENAME,
|
|
25
|
+
ROUND_MANIFEST_FILENAME,
|
|
26
|
+
ResumeError,
|
|
27
|
+
ResumePlan,
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _manifest_block_error(block_name: str, error: Exception) -> ResumeError:
|
|
32
|
+
return ResumeError(f"round manifest has malformed {block_name} block: {error}")
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _round_manifest_indicates_phase_failure(manifest: dict) -> bool:
|
|
36
|
+
"""Inspect a round's manifest.json contents and decide whether
|
|
37
|
+
the round had a phase failure (reviewer / synth / test-run /
|
|
38
|
+
producer subprocess or environment error).
|
|
39
|
+
|
|
40
|
+
NOT classified as a phase failure:
|
|
41
|
+
|
|
42
|
+
- ``test_run.outcome == "failed"`` — the test ran and exited
|
|
43
|
+
non-zero. That's a clean signal the loop terminator uses to
|
|
44
|
+
compute the round's exit code; not a phase failure.
|
|
45
|
+
- ``producer.outcome == "stalled"`` — the producer subprocess
|
|
46
|
+
exited cleanly without committing. Clean signal; the loop
|
|
47
|
+
terminator handles it as ``producer_stalled``.
|
|
48
|
+
|
|
49
|
+
Classified as a phase failure (and therefore drop-and-retry):
|
|
50
|
+
|
|
51
|
+
- Any reviewer ``outcome`` other than ``"success"``.
|
|
52
|
+
- ``synthesizer.outcome != "success"`` when the synth phase ran.
|
|
53
|
+
- ``test_run.outcome == "subprocess_error"`` when the test phase
|
|
54
|
+
ran.
|
|
55
|
+
- ``producer.outcome == "subprocess_error"`` when the producer
|
|
56
|
+
phase ran.
|
|
57
|
+
- ``producer.outcome == "escalated"`` — not an error, but a
|
|
58
|
+
deliberate checkpoint: the round re-runs from snapshot with the
|
|
59
|
+
operator's recorded decision. (Distinct from ``stalled``.)
|
|
60
|
+
- A BLOCKING ``checks[]`` entry with ``outcome ==
|
|
61
|
+
"subprocess_error"`` — mirrors
|
|
62
|
+
``verdict._compute_exit_code`` (→ exit 40 / ``REVIEWER_FAILURE``)
|
|
63
|
+
and ``verdict._classify_phase_failure`` (→
|
|
64
|
+
``"check_subprocess_error"``), an exit code that IS resume-eligible.
|
|
65
|
+
A blocking check that merely FAILED (→ exit 30) is a CLEAN NO-SHIP
|
|
66
|
+
signal like ``test_run == "failed"`` — NOT a phase failure.
|
|
67
|
+
|
|
68
|
+
The intent is "should this round be dropped and retried on resume?"
|
|
69
|
+
— for the error cases, because the original loop aborted there; for
|
|
70
|
+
escalation, because the operator's decision now lets the round
|
|
71
|
+
proceed. Either way we drop + retry rather than re-use partial state.
|
|
72
|
+
"""
|
|
73
|
+
# A before-dispatch refusal must be retried: the run refused without dispatching
|
|
74
|
+
# any reviewer, so the round is incomplete. A resumed attempt may succeed if the
|
|
75
|
+
# underlying issue (malformed headers, oversize diff, oversize prompt) is resolved.
|
|
76
|
+
# `refusal_reason` is the unified discriminator added in PR-h-field-01 round 1;
|
|
77
|
+
# `diff_filter_refusal_headers` is the legacy per-run artifact for diff_malformed
|
|
78
|
+
# runs written before that field existed — both are checked for backward compat.
|
|
79
|
+
_refusal_reason = manifest.get("refusal_reason")
|
|
80
|
+
if _refusal_reason in {"diff_malformed", "diff_too_large", "prompt_too_large"}:
|
|
81
|
+
return True
|
|
82
|
+
if manifest.get("diff_filter_refusal_headers") is not None:
|
|
83
|
+
return True
|
|
84
|
+
try:
|
|
85
|
+
for reviewer in manifest.get("reviewers") or []:
|
|
86
|
+
if reviewer["outcome"] != "success":
|
|
87
|
+
return True
|
|
88
|
+
except (KeyError, TypeError) as e:
|
|
89
|
+
raise _manifest_block_error("reviewers", e) from e
|
|
90
|
+
synth = manifest.get("synthesizer")
|
|
91
|
+
try:
|
|
92
|
+
if synth is not None and synth["outcome"] != "success":
|
|
93
|
+
return True
|
|
94
|
+
except (KeyError, TypeError) as e:
|
|
95
|
+
raise _manifest_block_error("synthesizer", e) from e
|
|
96
|
+
test_run = manifest.get("test_run")
|
|
97
|
+
try:
|
|
98
|
+
test_outcome = test_run["outcome"] if test_run is not None else None
|
|
99
|
+
if test_run is not None:
|
|
100
|
+
test_run["exit_code"]
|
|
101
|
+
if test_outcome not in ("passed", "failed", "subprocess_error"):
|
|
102
|
+
raise _manifest_block_error(
|
|
103
|
+
"test_run", ValueError(f"unexpected outcome {test_outcome!r}")
|
|
104
|
+
)
|
|
105
|
+
if test_outcome == "subprocess_error":
|
|
106
|
+
return True
|
|
107
|
+
except (KeyError, TypeError) as e:
|
|
108
|
+
raise _manifest_block_error("test_run", e) from e
|
|
109
|
+
# A test-worktree provisioning failure is an environment abort: test_run
|
|
110
|
+
# is null (the test never started) but test_skip_reason records the cause.
|
|
111
|
+
if manifest.get("test_skip_reason") == "test_worktree_error":
|
|
112
|
+
return True
|
|
113
|
+
producer = manifest.get("producer")
|
|
114
|
+
try:
|
|
115
|
+
producer_outcome = producer["outcome"] if producer is not None else None
|
|
116
|
+
if producer is not None:
|
|
117
|
+
producer["starting_sha"]
|
|
118
|
+
producer["ending_sha"]
|
|
119
|
+
if producer_outcome not in ("committed", "stalled", "subprocess_error", "escalated"):
|
|
120
|
+
raise _manifest_block_error(
|
|
121
|
+
"producer", ValueError(f"unexpected outcome {producer_outcome!r}")
|
|
122
|
+
)
|
|
123
|
+
except (KeyError, TypeError) as e:
|
|
124
|
+
raise _manifest_block_error("producer", e) from e
|
|
125
|
+
if producer_outcome == "subprocess_error":
|
|
126
|
+
return True
|
|
127
|
+
# a producer ESCALATION is not an error, but it IS a
|
|
128
|
+
# drop-and-retry trigger — the round re-runs from snapshot with the
|
|
129
|
+
# operator's recorded decision fed to the producer. (Distinct from
|
|
130
|
+
# ``stalled``, which is a clean terminal signal and is NOT retried.)
|
|
131
|
+
if producer_outcome == "escalated":
|
|
132
|
+
return True
|
|
133
|
+
# / §5 drift fix: a BLOCKING mechanical check whose subprocess
|
|
134
|
+
# could not run drives the round to exit 40 (REVIEWER_FAILURE), which
|
|
135
|
+
# is resume-eligible. Mirror verdict._classify_phase_failure's exact
|
|
136
|
+
# predicate so the two surfaces agree — without this the errored round
|
|
137
|
+
# is mis-read as completed cleanly and never re-run. A blocking check
|
|
138
|
+
# that merely FAILED (→ exit 30) is a clean NO-SHIP signal, NOT a
|
|
139
|
+
# phase failure (intentionally excluded, like test_run "failed").
|
|
140
|
+
try:
|
|
141
|
+
for check in manifest.get("checks") or []:
|
|
142
|
+
if is_blocking_check_subprocess_error(check["severity"], check["outcome"]):
|
|
143
|
+
return True
|
|
144
|
+
except (KeyError, TypeError) as e:
|
|
145
|
+
raise _manifest_block_error("checks", e) from e
|
|
146
|
+
return False
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def _reject_malformed_completed_round(manifest: dict, manifest_path: Path) -> None:
|
|
150
|
+
"""Validate the manifest-level fields a completed round's rehydration
|
|
151
|
+
relies on — ``snapshot.diff_present`` and each reviewer ``name``.
|
|
152
|
+
|
|
153
|
+
The plan walk calls this for a round that claims SHIP
|
|
154
|
+
(``round_exit_code == 0``) BEFORE the not-resumable refusal: a SHIPped
|
|
155
|
+
round is never rehydrated, so without this its structural corruption
|
|
156
|
+
would hide behind the generic "already SHIPped" message. NO-SHIP
|
|
157
|
+
completed rounds get the same checks (plus parsed.json validation) in
|
|
158
|
+
:func:`load_completed_round` when they are actually rehydrated, so this
|
|
159
|
+
helper stays manifest-only and does NOT require the parsed.json sidecar
|
|
160
|
+
files a refused SHIP round never persists. The field messages mirror
|
|
161
|
+
``load_completed_round`` so the surfaced error reads identically
|
|
162
|
+
regardless of which path caught the corruption. Raises
|
|
163
|
+
:class:`ResumeError` (→ exit 60) on a malformed field.
|
|
164
|
+
"""
|
|
165
|
+
snapshot = manifest.get("snapshot")
|
|
166
|
+
if isinstance(snapshot, dict):
|
|
167
|
+
diff_present = snapshot.get("diff_present", False)
|
|
168
|
+
if not isinstance(diff_present, bool):
|
|
169
|
+
raise ResumeError(
|
|
170
|
+
f"{manifest_path} has malformed snapshot block: "
|
|
171
|
+
f"snapshot.diff_present must be bool, got {diff_present!r}"
|
|
172
|
+
)
|
|
173
|
+
for reviewer in manifest.get("reviewers") or []:
|
|
174
|
+
name = reviewer.get("name") if isinstance(reviewer, dict) else None
|
|
175
|
+
if not isinstance(name, str) or not name:
|
|
176
|
+
raise ResumeError(
|
|
177
|
+
f"{manifest_path} has malformed reviewers block: "
|
|
178
|
+
f"reviewer name must be a non-empty string, got {name!r}"
|
|
179
|
+
)
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def read_resume_decision(run_dir: Path, resumed_round: int) -> str | None:
|
|
183
|
+
"""Resolve the operator decision to feed the resumed round's producer.
|
|
184
|
+
|
|
185
|
+
Returns ``None`` when the round being resumed did NOT escalate (a
|
|
186
|
+
normal environment-failure resume — there's no decision to inject).
|
|
187
|
+
|
|
188
|
+
When the round DID escalate (``producer.outcome == "escalated"`` in
|
|
189
|
+
its manifest), the operator must have recorded a decision in
|
|
190
|
+
``<run_dir>/decision.txt`` before resuming:
|
|
191
|
+
|
|
192
|
+
- decision present → returns its stripped text (fed to the producer
|
|
193
|
+
via the ``{operator_decision}`` prompt field).
|
|
194
|
+
- decision absent / blank → raises :class:`ResumeError` so the CLI
|
|
195
|
+
refuses the resume with a helpful message rather than re-running
|
|
196
|
+
the round only to escalate again.
|
|
197
|
+
|
|
198
|
+
Read BEFORE the resumed round's directory is dropped (the manifest
|
|
199
|
+
is gone after the drop).
|
|
200
|
+
"""
|
|
201
|
+
manifest_path = run_dir / f"round-{resumed_round}" / ROUND_MANIFEST_FILENAME
|
|
202
|
+
if not manifest_path.is_file():
|
|
203
|
+
return None
|
|
204
|
+
try:
|
|
205
|
+
manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
|
|
206
|
+
except json.JSONDecodeError:
|
|
207
|
+
return None
|
|
208
|
+
producer = manifest.get("producer")
|
|
209
|
+
if not (isinstance(producer, dict) and producer.get("outcome") == "escalated"):
|
|
210
|
+
return None
|
|
211
|
+
decision = read_operator_decision(run_dir)
|
|
212
|
+
if decision is None:
|
|
213
|
+
raise ResumeError(
|
|
214
|
+
f"run {run_dir.name} stopped for an operator decision — the "
|
|
215
|
+
f"round-{resumed_round} producer escalated a finding. Record your "
|
|
216
|
+
f"decision in {run_dir / OPERATOR_DECISION_FILENAME} before "
|
|
217
|
+
f"resuming (see decision-needed.md for the producer's case + "
|
|
218
|
+
f"options)."
|
|
219
|
+
)
|
|
220
|
+
return decision
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def _expected_sha_for_resumed_round(
|
|
224
|
+
*,
|
|
225
|
+
resumed_round: int,
|
|
226
|
+
starting_sha: str,
|
|
227
|
+
run_dir: Path,
|
|
228
|
+
) -> str:
|
|
229
|
+
"""Derive the snapshot SHA the resumed round should pin against.
|
|
230
|
+
|
|
231
|
+
Round 0 uses ``run-init.json::starting_sha``. Round N>0 reads
|
|
232
|
+
the prior round's manifest: if the prior producer committed,
|
|
233
|
+
use ``producer.ending_sha``; otherwise (no producer commit),
|
|
234
|
+
the branch never advanced, so use the prior round's
|
|
235
|
+
``snapshot.commit_sha``.
|
|
236
|
+
|
|
237
|
+
Raises :class:`ResumeError` when the round-(N-1) manifest is
|
|
238
|
+
missing or malformed — that's a degenerate state plan_resume
|
|
239
|
+
can't recover from.
|
|
240
|
+
"""
|
|
241
|
+
if resumed_round == 0:
|
|
242
|
+
return starting_sha
|
|
243
|
+
prior_round_dir = run_dir / f"round-{resumed_round - 1}"
|
|
244
|
+
prior_manifest_path = prior_round_dir / ROUND_MANIFEST_FILENAME
|
|
245
|
+
if not prior_manifest_path.is_file():
|
|
246
|
+
raise ResumeError(
|
|
247
|
+
f"resumed round {resumed_round} but prior round directory "
|
|
248
|
+
f"{prior_round_dir} has no {ROUND_MANIFEST_FILENAME} — "
|
|
249
|
+
f"cannot derive expected snapshot SHA"
|
|
250
|
+
)
|
|
251
|
+
try:
|
|
252
|
+
prior_manifest = json.loads(prior_manifest_path.read_text(encoding="utf-8"))
|
|
253
|
+
except json.JSONDecodeError as e:
|
|
254
|
+
raise ResumeError(f"prior round manifest at {prior_manifest_path} is malformed: {e}") from e
|
|
255
|
+
producer = prior_manifest.get("producer")
|
|
256
|
+
if producer is not None and producer.get("outcome") == "committed":
|
|
257
|
+
ending_sha = producer.get("ending_sha")
|
|
258
|
+
if isinstance(ending_sha, str) and ending_sha:
|
|
259
|
+
return ending_sha
|
|
260
|
+
# No producer commit (stalled, errored, or producer phase didn't
|
|
261
|
+
# run) → the branch was never advanced; resumed round snapshots
|
|
262
|
+
# from the prior round's snapshot SHA.
|
|
263
|
+
snapshot = prior_manifest.get("snapshot") or {}
|
|
264
|
+
commit_sha = snapshot.get("commit_sha")
|
|
265
|
+
if not isinstance(commit_sha, str) or not commit_sha:
|
|
266
|
+
raise ResumeError(
|
|
267
|
+
f"prior round manifest at {prior_manifest_path} has no recoverable snapshot.commit_sha"
|
|
268
|
+
)
|
|
269
|
+
return commit_sha
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def plan_resume(
|
|
273
|
+
repo_root: Path,
|
|
274
|
+
run_dir: Path,
|
|
275
|
+
max_rounds_override: int | None = None,
|
|
276
|
+
) -> ResumePlan:
|
|
277
|
+
"""Build a :class:`ResumePlan` for the given run directory.
|
|
278
|
+
|
|
279
|
+
Reads ``run-init.json`` to recover the original-run context,
|
|
280
|
+
walks ``round-0/``, ``round-1/``, ... to find the first
|
|
281
|
+
incomplete round, and derives the expected snapshot SHA for
|
|
282
|
+
that round.
|
|
283
|
+
|
|
284
|
+
Args:
|
|
285
|
+
repo_root: Repo root (used to make ``pr_doc_path``
|
|
286
|
+
absolute when run-init.json recorded a relative path).
|
|
287
|
+
run_dir: The ``<runs_root>/<run_id>/`` directory.
|
|
288
|
+
max_rounds_override: When supplied (e.g. from the CLI's
|
|
289
|
+
``--resume --max-rounds N``), the effective walk range
|
|
290
|
+
is ``max(original_max_rounds, max_rounds_override)``.
|
|
291
|
+
This lets ``--resume --max-rounds 2`` continue at
|
|
292
|
+
round 1 even when the original run used
|
|
293
|
+
``max_rounds=1`` (all 1 rounds completed cleanly but
|
|
294
|
+
the loop-manifest recorded exit 25/40/60/70).
|
|
295
|
+
|
|
296
|
+
Returns:
|
|
297
|
+
A :class:`ResumePlan` populated from on-disk artifacts.
|
|
298
|
+
|
|
299
|
+
Raises:
|
|
300
|
+
ResumeError: If ``run-init.json`` is missing or malformed,
|
|
301
|
+
or if no incomplete round can be identified AND the
|
|
302
|
+
effective round cap is exhausted (degenerate state).
|
|
303
|
+
"""
|
|
304
|
+
if not run_dir.is_dir():
|
|
305
|
+
raise ResumeError(f"run directory does not exist: {run_dir}")
|
|
306
|
+
_refuse_if_blockers_all_deactivated(run_dir)
|
|
307
|
+
run_init_path = run_dir / RUN_INIT_FILENAME
|
|
308
|
+
if not run_init_path.is_file():
|
|
309
|
+
raise ResumeError(f"{RUN_INIT_FILENAME} missing in {run_dir} — cannot resume")
|
|
310
|
+
try:
|
|
311
|
+
run_init = json.loads(run_init_path.read_text(encoding="utf-8"))
|
|
312
|
+
except json.JSONDecodeError as e:
|
|
313
|
+
raise ResumeError(f"{run_init_path} is malformed: {e}") from e
|
|
314
|
+
|
|
315
|
+
try:
|
|
316
|
+
max_rounds = int(run_init["max_rounds"])
|
|
317
|
+
starting_sha = str(run_init["starting_sha"])
|
|
318
|
+
syncade_version = str(run_init["syncade_version"])
|
|
319
|
+
pr_doc_str = str(run_init["pr_doc_path"])
|
|
320
|
+
except (KeyError, TypeError, ValueError) as e:
|
|
321
|
+
raise ResumeError(f"{run_init_path} is missing a required field: {e}") from e
|
|
322
|
+
# Validate starting_sha on read — mirrors the live-git HEAD check in
|
|
323
|
+
# resume_load._current_head_sha. A malformed value would otherwise only
|
|
324
|
+
# surface downstream as a TreeDriftError (kind="sha") when the real HEAD
|
|
325
|
+
# fails to match it, masking the real cause (corrupt run-init.json).
|
|
326
|
+
if not is_full_git_object_id(starting_sha):
|
|
327
|
+
raise ResumeError(
|
|
328
|
+
f"{run_init_path} has a malformed starting_sha {starting_sha!r}; "
|
|
329
|
+
f"expected a full SHA-1/SHA-256 object ID"
|
|
330
|
+
)
|
|
331
|
+
operator_branch = run_init.get("operator_branch")
|
|
332
|
+
base_ref = run_init.get("base_ref")
|
|
333
|
+
base_oid = run_init.get("base_oid")
|
|
334
|
+
if base_oid is not None and (
|
|
335
|
+
not isinstance(base_oid, str) or not is_full_git_object_id(base_oid)
|
|
336
|
+
):
|
|
337
|
+
raise ResumeError(
|
|
338
|
+
f"{run_init_path} has a malformed base_oid {base_oid!r}; "
|
|
339
|
+
f"expected a full SHA-1/SHA-256 object ID or null"
|
|
340
|
+
)
|
|
341
|
+
# A legacy run (pre-review-identity fix) recorded base_ref but never pinned
|
|
342
|
+
# base_oid. Resuming it would re-resolve the symbolic ref under three-dot
|
|
343
|
+
# semantics, producing a different diff than the earlier rounds reviewed.
|
|
344
|
+
# Refuse so the operator starts a fresh run with a stable base instead.
|
|
345
|
+
if base_oid is None and base_ref is not None:
|
|
346
|
+
raise ResumeError(
|
|
347
|
+
f"{run_init_path} has no base_oid (recorded before the review-identity "
|
|
348
|
+
f"fix). Resuming it would review a different diff than earlier rounds: "
|
|
349
|
+
f"earlier rounds used {base_ref!r} with two-dot semantics; a resume "
|
|
350
|
+
f"would use three-dot semantics instead. Start a fresh run instead: "
|
|
351
|
+
f"syncade --base {base_ref!r} <PR_DOC>"
|
|
352
|
+
)
|
|
353
|
+
if operator_branch is not None and not isinstance(operator_branch, str):
|
|
354
|
+
raise ResumeError(
|
|
355
|
+
f"{run_init_path} has malformed operator_branch "
|
|
356
|
+
f"({operator_branch!r}); expected str or null"
|
|
357
|
+
)
|
|
358
|
+
|
|
359
|
+
# Resolve pr_doc_path relative to repo_root when the recorded
|
|
360
|
+
# path is relative (the orchestrator persists str(path); if the
|
|
361
|
+
# original run was invoked via a relative path the absolute
|
|
362
|
+
# form would differ across machines but the relative form is
|
|
363
|
+
# repo-portable).
|
|
364
|
+
pr_doc_path = Path(pr_doc_str)
|
|
365
|
+
if not pr_doc_path.is_absolute():
|
|
366
|
+
pr_doc_path = (repo_root / pr_doc_path).resolve()
|
|
367
|
+
|
|
368
|
+
# When the CLI passes --max-rounds N, the effective walk range
|
|
369
|
+
# extends to cover the extra rounds the operator wants to add.
|
|
370
|
+
effective_max_rounds = (
|
|
371
|
+
max(max_rounds, max_rounds_override) if max_rounds_override is not None else max_rounds
|
|
372
|
+
)
|
|
373
|
+
|
|
374
|
+
# Walk rounds 0..effective_max_rounds-1 to find the first incomplete one.
|
|
375
|
+
completed_rounds: list[int] = []
|
|
376
|
+
resumed_round: int | None = None
|
|
377
|
+
budget_aborted_before_producer_round: int | None = None
|
|
378
|
+
for round_idx in range(effective_max_rounds):
|
|
379
|
+
round_dir = run_dir / f"round-{round_idx}"
|
|
380
|
+
if not round_dir.is_dir():
|
|
381
|
+
resumed_round = round_idx
|
|
382
|
+
break
|
|
383
|
+
manifest_path = round_dir / ROUND_MANIFEST_FILENAME
|
|
384
|
+
if not manifest_path.is_file():
|
|
385
|
+
# Round started but never persisted its manifest —
|
|
386
|
+
# typical interrupted-mid-round case.
|
|
387
|
+
resumed_round = round_idx
|
|
388
|
+
break
|
|
389
|
+
try:
|
|
390
|
+
manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
|
|
391
|
+
except json.JSONDecodeError:
|
|
392
|
+
# Malformed manifest → drop + retry the round.
|
|
393
|
+
resumed_round = round_idx
|
|
394
|
+
break
|
|
395
|
+
if _round_manifest_indicates_phase_failure(manifest):
|
|
396
|
+
resumed_round = round_idx
|
|
397
|
+
break
|
|
398
|
+
# A SHIPped round (round_exit_code == SUCCESS) terminates the run:
|
|
399
|
+
# the producer's commit was accepted and the loop ended. Such a run
|
|
400
|
+
# is only "resumable" when the loop crashed after this round
|
|
401
|
+
# persisted but before the terminator wrote its exit-0 loop-manifest.
|
|
402
|
+
# Resuming would re-review the un-advanced tree and could flip a
|
|
403
|
+
# legit SHIP to NO-SHIP and re-advance the branch. Refuse — the
|
|
404
|
+
# operator's next move is a fresh run.
|
|
405
|
+
if manifest.get("round_exit_code") == SUCCESS:
|
|
406
|
+
# A SHIPped round is never rehydrated, so its manifest fields are
|
|
407
|
+
# otherwise never validated. Reject a structurally-malformed
|
|
408
|
+
# manifest first (exit 60 naming the specific corruption) — it
|
|
409
|
+
# can't be trusted to have actually SHIPped, and the generic
|
|
410
|
+
# refusal below would mask the real problem.
|
|
411
|
+
_reject_malformed_completed_round(manifest, manifest_path)
|
|
412
|
+
raise ResumeError(
|
|
413
|
+
f"round-{round_idx} in {run_dir} already SHIPped "
|
|
414
|
+
f"(round_exit_code == {SUCCESS}); the run terminated with a "
|
|
415
|
+
f"SHIP and is not resumable. Start a fresh run."
|
|
416
|
+
)
|
|
417
|
+
# A non-final NO-SHIP round with no producer block means the
|
|
418
|
+
# producer phase never ran. Two sub-cases:
|
|
419
|
+
#
|
|
420
|
+
# (a) Budget abort: the loop manifest exists and records
|
|
421
|
+
# termination_reason == "budget_exceeded". The review bundle
|
|
422
|
+
# (reviewers + synth + test) is complete; only the producer
|
|
423
|
+
# was skipped because the pre-producer budget check tripped.
|
|
424
|
+
# Treat as completed (rehydrate the review bundle) and set
|
|
425
|
+
# budget_aborted_before_producer_round so the loop dispatches
|
|
426
|
+
# only the producer under the fresh budget tally.
|
|
427
|
+
#
|
|
428
|
+
# (b) Genuine interrupt: the process was killed between the first
|
|
429
|
+
# persist_round_manifest call and the producer dispatch. The
|
|
430
|
+
# loop manifest is absent or doesn't say budget_exceeded.
|
|
431
|
+
# Drop and retry the whole round.
|
|
432
|
+
if (
|
|
433
|
+
manifest.get("producer") is None
|
|
434
|
+
and manifest.get("round_exit_code") == FINDINGS_PRESENT
|
|
435
|
+
and round_idx < effective_max_rounds - 1
|
|
436
|
+
):
|
|
437
|
+
loop_manifest_path = run_dir / LOOP_MANIFEST_FILENAME
|
|
438
|
+
_budget_aborted = False
|
|
439
|
+
if loop_manifest_path.is_file():
|
|
440
|
+
try:
|
|
441
|
+
lm = json.loads(loop_manifest_path.read_text(encoding="utf-8"))
|
|
442
|
+
_budget_aborted = lm.get("termination_reason") == "budget_exceeded"
|
|
443
|
+
except (json.JSONDecodeError, OSError):
|
|
444
|
+
pass
|
|
445
|
+
if _budget_aborted:
|
|
446
|
+
# (a) review bundle complete — rehydrate it, resume at producer only
|
|
447
|
+
completed_rounds.append(round_idx)
|
|
448
|
+
resumed_round = round_idx
|
|
449
|
+
budget_aborted_before_producer_round = round_idx
|
|
450
|
+
else:
|
|
451
|
+
# (b) genuine interrupt — drop and retry whole round
|
|
452
|
+
resumed_round = round_idx
|
|
453
|
+
break
|
|
454
|
+
completed_rounds.append(round_idx)
|
|
455
|
+
|
|
456
|
+
if resumed_round is None:
|
|
457
|
+
# All effective_max_rounds rounds completed cleanly but
|
|
458
|
+
# eligibility got us here — the loop terminator must have
|
|
459
|
+
# aborted after the final round's persistence (rare: e.g.
|
|
460
|
+
# loop-manifest write failed). Per the brief:
|
|
461
|
+
# resumed_round = N+1 if not at cap, else degenerate state.
|
|
462
|
+
if len(completed_rounds) < effective_max_rounds:
|
|
463
|
+
# Defensive — shouldn't happen because the walk above
|
|
464
|
+
# adds every successful round to completed_rounds and
|
|
465
|
+
# the iteration count is effective_max_rounds, but kept
|
|
466
|
+
# explicit.
|
|
467
|
+
resumed_round = len(completed_rounds)
|
|
468
|
+
else:
|
|
469
|
+
raise ResumeError(
|
|
470
|
+
f"all {effective_max_rounds} rounds in {run_dir} completed "
|
|
471
|
+
f"cleanly per their per-round manifests, but the run "
|
|
472
|
+
f"is eligible to resume (loop-manifest absent or "
|
|
473
|
+
f"final_exit_code in {{25, 40, 60, 70}}). This is a "
|
|
474
|
+
f"degenerate state — the loop terminator aborted after "
|
|
475
|
+
f"the final round's persistence. Inspect "
|
|
476
|
+
f"{run_dir / LOOP_MANIFEST_FILENAME} (if present) "
|
|
477
|
+
f"and start a fresh run rather than resuming."
|
|
478
|
+
)
|
|
479
|
+
|
|
480
|
+
expected_sha = _expected_sha_for_resumed_round(
|
|
481
|
+
resumed_round=resumed_round,
|
|
482
|
+
starting_sha=starting_sha,
|
|
483
|
+
run_dir=run_dir,
|
|
484
|
+
)
|
|
485
|
+
|
|
486
|
+
return ResumePlan(
|
|
487
|
+
run_id=run_dir.name,
|
|
488
|
+
run_dir=run_dir,
|
|
489
|
+
pr_doc_path=pr_doc_path,
|
|
490
|
+
operator_branch=operator_branch,
|
|
491
|
+
expected_sha=expected_sha,
|
|
492
|
+
expected_branch=operator_branch,
|
|
493
|
+
resumed_round=resumed_round,
|
|
494
|
+
completed_rounds=completed_rounds,
|
|
495
|
+
max_rounds=max_rounds,
|
|
496
|
+
syncade_version=syncade_version,
|
|
497
|
+
config_snapshot_path=run_init_path,
|
|
498
|
+
base_ref=base_ref,
|
|
499
|
+
base_oid=base_oid,
|
|
500
|
+
budget_aborted_before_producer_round=budget_aborted_before_producer_round,
|
|
501
|
+
)
|
|
502
|
+
|
|
503
|
+
|
|
504
|
+
def _refuse_if_blockers_all_deactivated(run_dir: Path) -> None:
|
|
505
|
+
"""Refuse to resume a run that ended because every reviewer blocker was
|
|
506
|
+
deactivated (PR-h-01 increment D).
|
|
507
|
+
|
|
508
|
+
There is nothing to continue: no producer ran, no blocker is active for one
|
|
509
|
+
to fix, and the tree is unchanged — so a resume re-reviews byte-identical
|
|
510
|
+
code and deterministically reproduces the same exit 10, at the cost of a
|
|
511
|
+
full reviewer panel plus the judge. Worse, ``decision.txt`` is silently
|
|
512
|
+
ignored on this path (the resume decision reader is keyed to an escalated
|
|
513
|
+
PRODUCER round), so an operator following the old producer-escalation
|
|
514
|
+
instructions would pay for a round that cannot use their answer.
|
|
515
|
+
|
|
516
|
+
The question this state poses — was the synthesizer right to rule those
|
|
517
|
+
blockers out? — is answered by reading, then either accepting the round or
|
|
518
|
+
fixing the concern and starting a fresh run.
|
|
519
|
+
"""
|
|
520
|
+
loop_manifest_path = run_dir / LOOP_MANIFEST_FILENAME
|
|
521
|
+
if not loop_manifest_path.is_file():
|
|
522
|
+
# Manifest missing: usually interrupted, but the blockers-all-deactivated
|
|
523
|
+
# decision-needed.md is written BEFORE the manifest. Check the marker so
|
|
524
|
+
# a partial-finalization window doesn't reopen the non-resumable path.
|
|
525
|
+
from syncade.orchestrator.resume_target import _decision_needed_is_deactivated_shape
|
|
526
|
+
|
|
527
|
+
if not _decision_needed_is_deactivated_shape(run_dir):
|
|
528
|
+
return
|
|
529
|
+
raise ResumeError(
|
|
530
|
+
"this run ended at exit 10 because two or more reviewers each raised a "
|
|
531
|
+
"blocker and the synthesizer deactivated all of them — there is nothing "
|
|
532
|
+
"to resume. No producer ran, no blocker is active for one to fix, and "
|
|
533
|
+
"the tree has not changed, so resuming would re-run the full panel and "
|
|
534
|
+
"reproduce the same result. Read decision-needed.md in the run "
|
|
535
|
+
"directory: it quotes what each reviewer actually said next to what the "
|
|
536
|
+
"synthesizer did with it. If the synthesizer was right, the round is "
|
|
537
|
+
"effectively a SHIP; if not, fix the concern and start a fresh run."
|
|
538
|
+
)
|
|
539
|
+
try:
|
|
540
|
+
manifest = json.loads(loop_manifest_path.read_text(encoding="utf-8"))
|
|
541
|
+
except (json.JSONDecodeError, OSError):
|
|
542
|
+
return
|
|
543
|
+
if manifest.get("termination_reason") != "blockers_all_deactivated":
|
|
544
|
+
return
|
|
545
|
+
raise ResumeError(
|
|
546
|
+
"this run ended at exit 10 because two or more reviewers each raised a "
|
|
547
|
+
"blocker and the synthesizer deactivated all of them — there is nothing "
|
|
548
|
+
"to resume. No producer ran, no blocker is active for one to fix, and "
|
|
549
|
+
"the tree has not changed, so resuming would re-run the full panel and "
|
|
550
|
+
"reproduce the same result. Read decision-needed.md in the run "
|
|
551
|
+
"directory: it quotes what each reviewer actually said next to what the "
|
|
552
|
+
"synthesizer did with it. If the synthesizer was right, the round is "
|
|
553
|
+
"effectively a SHIP; if not, fix the concern and start a fresh run."
|
|
554
|
+
)
|