syncade 0.6.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (177) hide show
  1. syncade/__init__.py +3 -0
  2. syncade/__main__.py +6 -0
  3. syncade/adapters/__init__.py +0 -0
  4. syncade/adapters/anthropic.py +457 -0
  5. syncade/adapters/base.py +221 -0
  6. syncade/adapters/fake.py +73 -0
  7. syncade/adapters/fake_common.py +29 -0
  8. syncade/adapters/fake_producer_audit_draft.py +460 -0
  9. syncade/adapters/fake_reviewer_synth.py +310 -0
  10. syncade/adapters/openai.py +484 -0
  11. syncade/adapters/openai_parsing.py +119 -0
  12. syncade/adapters/producer.py +221 -0
  13. syncade/adapters/producer_anthropic.py +300 -0
  14. syncade/adapters/producer_openai.py +226 -0
  15. syncade/adapters/registry.py +81 -0
  16. syncade/auth_check.py +554 -0
  17. syncade/auth_preflight.py +342 -0
  18. syncade/base_resolution.py +214 -0
  19. syncade/billing.py +141 -0
  20. syncade/checks_config.py +113 -0
  21. syncade/cli/__init__.py +546 -0
  22. syncade/cli/auth_gate.py +59 -0
  23. syncade/cli/config_keys.py +135 -0
  24. syncade/cli/config_list.py +82 -0
  25. syncade/cli/config_menu_rows.py +166 -0
  26. syncade/cli/config_mode.py +609 -0
  27. syncade/cli/config_overrides.py +122 -0
  28. syncade/cli/config_tui.py +476 -0
  29. syncade/cli/doctor_mode.py +72 -0
  30. syncade/cli/gc_mode.py +109 -0
  31. syncade/cli/install_skill.py +514 -0
  32. syncade/cli/metrics_mode.py +363 -0
  33. syncade/cli/modes.py +573 -0
  34. syncade/cli/parser.py +450 -0
  35. syncade/cli/parser_types.py +137 -0
  36. syncade/cli/paths.py +38 -0
  37. syncade/cli/preflight_paths.py +90 -0
  38. syncade/cli/resolve.py +116 -0
  39. syncade/cli/resume_mode.py +324 -0
  40. syncade/cli/toml_writer.py +410 -0
  41. syncade/cli/validate.py +421 -0
  42. syncade/config.py +478 -0
  43. syncade/config_auth.py +310 -0
  44. syncade/config_cold.py +209 -0
  45. syncade/config_gc.py +55 -0
  46. syncade/config_loader.py +182 -0
  47. syncade/config_loop.py +282 -0
  48. syncade/config_producer.py +222 -0
  49. syncade/config_retry.py +49 -0
  50. syncade/config_types.py +59 -0
  51. syncade/diff_filter.py +437 -0
  52. syncade/dispatcher.py +571 -0
  53. syncade/doctor.py +425 -0
  54. syncade/doctor_env.py +218 -0
  55. syncade/doctor_preview.py +524 -0
  56. syncade/doctor_types.py +28 -0
  57. syncade/exit_codes.py +82 -0
  58. syncade/findings.py +242 -0
  59. syncade/findings_json.py +456 -0
  60. syncade/gc.py +211 -0
  61. syncade/gc_execute.py +372 -0
  62. syncade/gc_protection.py +129 -0
  63. syncade/gc_types.py +50 -0
  64. syncade/gc_worktrees.py +200 -0
  65. syncade/git_object_id.py +12 -0
  66. syncade/git_preconditions.py +389 -0
  67. syncade/logging.py +289 -0
  68. syncade/metrics/__init__.py +32 -0
  69. syncade/metrics/aggregate.py +550 -0
  70. syncade/metrics/schema.py +221 -0
  71. syncade/orchestrator/__init__.py +61 -0
  72. syncade/orchestrator/_runs_dir.py +24 -0
  73. syncade/orchestrator/branch_advance.py +165 -0
  74. syncade/orchestrator/branch_guard.py +98 -0
  75. syncade/orchestrator/budget.py +107 -0
  76. syncade/orchestrator/escalation_coverage.py +81 -0
  77. syncade/orchestrator/loop.py +611 -0
  78. syncade/orchestrator/loop_dispatch_check.py +112 -0
  79. syncade/orchestrator/loop_finalize.py +404 -0
  80. syncade/orchestrator/loop_preflight.py +131 -0
  81. syncade/orchestrator/loop_resume.py +91 -0
  82. syncade/orchestrator/loop_rmtree.py +70 -0
  83. syncade/orchestrator/loop_round_step.py +599 -0
  84. syncade/orchestrator/prior_round.py +336 -0
  85. syncade/orchestrator/producer_phase.py +169 -0
  86. syncade/orchestrator/results.py +306 -0
  87. syncade/orchestrator/resume.py +96 -0
  88. syncade/orchestrator/resume_load.py +483 -0
  89. syncade/orchestrator/resume_plan.py +554 -0
  90. syncade/orchestrator/resume_target.py +215 -0
  91. syncade/orchestrator/resume_types.py +182 -0
  92. syncade/orchestrator/reviewer_template_failure.py +99 -0
  93. syncade/orchestrator/round.py +573 -0
  94. syncade/orchestrator/round_checks.py +91 -0
  95. syncade/orchestrator/round_no_changes.py +369 -0
  96. syncade/orchestrator/round_predispatch.py +212 -0
  97. syncade/orchestrator/verdict.py +279 -0
  98. syncade/persistence/__init__.py +189 -0
  99. syncade/persistence/_atomic.py +33 -0
  100. syncade/persistence/_clusters.py +70 -0
  101. syncade/persistence/_findings_verdict.py +201 -0
  102. syncade/persistence/_markdown.py +286 -0
  103. syncade/persistence/_validation.py +37 -0
  104. syncade/persistence/checks.py +249 -0
  105. syncade/persistence/decision_needed.py +289 -0
  106. syncade/persistence/findings_md.py +389 -0
  107. syncade/persistence/handoff.py +389 -0
  108. syncade/persistence/handoff_classify.py +196 -0
  109. syncade/persistence/last_reviewed.py +67 -0
  110. syncade/persistence/loop_manifest.py +165 -0
  111. syncade/persistence/loop_summary.py +352 -0
  112. syncade/persistence/loop_summary_text.py +428 -0
  113. syncade/persistence/producer.py +250 -0
  114. syncade/persistence/reviewer.py +198 -0
  115. syncade/persistence/round_manifest.py +238 -0
  116. syncade/persistence/run_init.py +153 -0
  117. syncade/persistence/run_summary.py +585 -0
  118. syncade/persistence/run_summary_next_steps.py +443 -0
  119. syncade/persistence/synth.py +242 -0
  120. syncade/persistence/test_run.py +152 -0
  121. syncade/presets.py +36 -0
  122. syncade/pricing_config.py +72 -0
  123. syncade/process.py +600 -0
  124. syncade/producer.py +189 -0
  125. syncade/producer_attempt.py +463 -0
  126. syncade/producer_escalation.py +146 -0
  127. syncade/producer_git.py +199 -0
  128. syncade/producer_result.py +205 -0
  129. syncade/prompts.py +448 -0
  130. syncade/prompts_loader.py +238 -0
  131. syncade/retry.py +159 -0
  132. syncade/run_inputs.py +40 -0
  133. syncade/run_status.py +198 -0
  134. syncade/selfcheck.py +471 -0
  135. syncade/skills/claude/README.md +221 -0
  136. syncade/skills/claude/SKILL.md +625 -0
  137. syncade/skills/codex/README.md +116 -0
  138. syncade/skills/codex/SKILL.md +574 -0
  139. syncade/snapshot.py +598 -0
  140. syncade/spec_audit.py +437 -0
  141. syncade/spec_audit_schema.py +190 -0
  142. syncade/spec_draft.py +423 -0
  143. syncade/spec_source.py +135 -0
  144. syncade/synthesis.py +428 -0
  145. syncade/synthesis_clusters.py +203 -0
  146. syncade/synthesis_repair.py +230 -0
  147. syncade/synthesis_schema.py +65 -0
  148. syncade/synthesizer/__init__.py +38 -0
  149. syncade/synthesizer/constants.py +33 -0
  150. syncade/synthesizer/driver.py +531 -0
  151. syncade/synthesizer/rendering.py +63 -0
  152. syncade/synthesizer/result.py +73 -0
  153. syncade/synthesizer/validation.py +421 -0
  154. syncade/synthesizer/workspace.py +208 -0
  155. syncade/templates/presets/balanced.toml +13 -0
  156. syncade/templates/presets/cheap.toml +12 -0
  157. syncade/templates/presets/thorough.toml +9 -0
  158. syncade/templates/producer.md +231 -0
  159. syncade/templates/reviewer.md +279 -0
  160. syncade/templates/reviewer_adversarial.md +164 -0
  161. syncade/templates/reviewer_codex.md +165 -0
  162. syncade/templates/spec_audit.md +168 -0
  163. syncade/templates/spec_draft.md +62 -0
  164. syncade/templates/synthesizer.md +204 -0
  165. syncade/test_runner.py +476 -0
  166. syncade/test_runner_classify.py +98 -0
  167. syncade/transcript.py +150 -0
  168. syncade/usage.py +407 -0
  169. syncade/worktree.py +497 -0
  170. syncade/worktree_env.py +133 -0
  171. syncade/worktree_paths.py +139 -0
  172. syncade-0.6.2.dist-info/METADATA +314 -0
  173. syncade-0.6.2.dist-info/RECORD +177 -0
  174. syncade-0.6.2.dist-info/WHEEL +5 -0
  175. syncade-0.6.2.dist-info/entry_points.txt +2 -0
  176. syncade-0.6.2.dist-info/licenses/LICENSE +202 -0
  177. syncade-0.6.2.dist-info/top_level.txt +1 -0
@@ -0,0 +1,611 @@
1
+ """Top-level multi-round review loop driver."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Callable
6
+ from datetime import UTC, datetime
7
+ from pathlib import Path
8
+ from typing import TYPE_CHECKING
9
+
10
+ from syncade import __version__, run_status
11
+ from syncade.adapters.base import ReviewerAdapter
12
+ from syncade.adapters.producer import ProducerAdapter
13
+ from syncade.config import SyncadeConfig
14
+ from syncade.exit_codes import BUDGET_EXCEEDED, SUCCESS, WORKTREE_ERROR
15
+ from syncade.logging import Logger, _approaching_budget_line
16
+ from syncade.persistence import persist_run_init
17
+ from syncade.persistence._atomic import atomic_write_text
18
+ from syncade.run_inputs import validate_run_inputs
19
+ from syncade.snapshot import SnapshotError, discover_repo_root, take_snapshot
20
+ from syncade.worktree import WorktreeError, WorktreeManager, generate_run_id
21
+
22
+ from ._runs_dir import _ensure_runs_gitignore
23
+ from .budget import approaching_budget, over_budget, producer_only_usages, round_usages
24
+ from .loop_dispatch_check import _diff_will_dispatch as _diff_will_dispatch
25
+ from .loop_finalize import _finalize_run
26
+ from .loop_preflight import run_preflight
27
+ from .loop_resume import _rehydrate_resume_state
28
+ from .loop_rmtree import _safe_resume_rmtree
29
+ from .loop_round_step import _run_round_step
30
+ from .results import RoundArtifacts, RoundResult, RunResult, TerminationReason
31
+ from .resume import ResumePlan, check_tree_drift
32
+ from .resume_types import ResumeError
33
+
34
+ if TYPE_CHECKING:
35
+ from syncade.usage import Usage
36
+
37
+
38
+ def _autoprune_old_transcripts(
39
+ repo_root: Path, logger: Logger, *, keep: int, max_age_days: int
40
+ ) -> None:
41
+ """Keep ``.syncade/runs/`` bounded without anyone remembering ``syncade --gc``.
42
+
43
+ **Never fails a review.** Auto-prune is housekeeping; a review is the product. Any
44
+ failure here — unreadable dir, permissions, a corpus mid-write by a concurrent
45
+ syncade — is swallowed and logged, never raised. The alternative (a disk-cleanup
46
+ error aborting an expensive multi-round loop) is strictly worse than a fat
47
+ ``.syncade/``.
48
+
49
+ **Quiet unless it did something.** A no-op prune prints nothing; the operator's
50
+ pane is for the review, not for housekeeping that freed zero bytes.
51
+
52
+ The import is function-local on purpose — the same reason
53
+ :func:`_safe_resume_rmtree` defers its GC imports. ``syncade.gc`` reaches
54
+ ``gc_protection`` → ``orchestrator.resume`` → ``orchestrator/__init__`` → this
55
+ module, so importing it at module scope makes ``import syncade.gc`` fail with a
56
+ partially-initialized-module error. The CLI happened to survive that (it enters
57
+ through a different module first) and the whole test suite happened to survive it
58
+ (pytest imports ``orchestrator`` first) — which is exactly why it has to be
59
+ deferred rather than trusted.
60
+ """
61
+ from syncade.gc import autoprune_transcripts
62
+
63
+ try:
64
+ report = autoprune_transcripts(repo_root, keep=keep, max_age_days=max_age_days)
65
+ except Exception as exc: # noqa: BLE001 — housekeeping must never abort a review
66
+ logger.warning(f"auto-prune skipped ({type(exc).__name__}: {exc})")
67
+ return
68
+ # Report on runs SLIMMED, not bytes freed — the same distinction execute_gc keys
69
+ # on. A run whose transcripts are all zero-byte is still mutated, and gating the
70
+ # log on bytes would leave the loop silent about work it actually did.
71
+ if report.runs_slimmed:
72
+ mb = report.bytes_freed / (1024 * 1024)
73
+ logger.event(
74
+ f"auto-pruned {len(report.runs_slimmed)} old run(s) — "
75
+ f"{mb:.1f} MB of transcripts freed (run history kept)"
76
+ )
77
+
78
+
79
+ def run_review(
80
+ *,
81
+ repo_root: Path,
82
+ pr_doc_path: Path,
83
+ config: SyncadeConfig,
84
+ base_ref: str | None = None,
85
+ timeout_seconds: float | None = None,
86
+ logger: Logger | None = None,
87
+ adapter_factory: Callable[[str], ReviewerAdapter] | None = None,
88
+ synthesizer_adapter: ReviewerAdapter | None = None,
89
+ producer_adapter: ProducerAdapter | None = None,
90
+ worktree_base: Path | None = None,
91
+ force_dirty: bool = False,
92
+ two_dot: bool = False,
93
+ resume_plan: ResumePlan | None = None,
94
+ force_drift: bool = False,
95
+ operator_decision: str | None = None,
96
+ pr_doc_artifact_name: str | None = None,
97
+ allow_default_branch: bool = False,
98
+ ) -> RunResult:
99
+ """Execute a multi-round review loop against ``repo_root``.
100
+
101
+ For ``config.loop.max_rounds == 1``, run one round without
102
+ provisioning a producer subprocess.
103
+
104
+ For ``max_rounds > 1``, wraps the per-round pipeline in a loop:
105
+ each iteration runs snapshot → reviewers → synth → optional test
106
+ → verdict → (if NO-SHIP and rounds remain) producer → branch
107
+ advance → next round. The loop terminates as soon as:
108
+
109
+ - **SHIP** at any round → exit 0 with
110
+ ``termination_reason="ship"``.
111
+ - **Max rounds reached** (last round was NO-SHIP) → exit 20
112
+ with ``termination_reason="max_rounds_reached"``.
113
+ - **Producer stall** (subprocess clean but no commit, OR an
114
+ escalation that does not cover every active blocker) → exit 30
115
+ with ``termination_reason="producer_stalled"``.
116
+ - **Decision needed** (the producer escalated and its
117
+ ``finding_indices`` covered every active blocker) → exit
118
+ 10 with ``termination_reason="decision_needed"`` +
119
+ ``decision-needed.md``.
120
+ - **Budget exceeded** (token or dollar ceiling hit mid-loop) →
121
+ exit 25 with ``termination_reason="budget_exceeded"``.
122
+ - **Provider usage limit** (the account's quota window is empty, so
123
+ no actor can be dispatched) → also exit 25, but with
124
+ ``termination_reason="provider_usage_limit"``: same resumable
125
+ stop, a cause the operator did not configure and cannot raise.
126
+ - **Subprocess / parse / worktree / config error** → exit
127
+ 40/50/60/70 with the appropriate categorical reason.
128
+
129
+ Args:
130
+ repo_root: Path inside the git repo. Resolved to the actual
131
+ git root via
132
+ :func:`~syncade.snapshot.discover_repo_root` before any
133
+ side effects so ``.syncade/`` artifacts land at the repo
134
+ root regardless of invocation cwd.
135
+ pr_doc_path: Path to the PR doc; substituted into both the
136
+ reviewer + producer prompts.
137
+ config: The loaded :class:`~syncade.config.SyncadeConfig`.
138
+ ``config.loop.max_rounds`` drives loop iteration count;
139
+ ``config.loop.test_command`` (if set) drives the per-
140
+ round test re-run leg; ``config.producer`` drives the
141
+ producer adapter config (provider / model / thinking /
142
+ permissions / timeout).
143
+ base_ref: Optional git ref the diff is rendered against on
144
+ every round. ``None`` produces the no-diff sentinel and
145
+ the reviewers operate against the full HEAD state.
146
+ timeout_seconds: Per-reviewer wall-clock timeout. Falls back
147
+ to ``config.loop.timeout_seconds``. Same value is used
148
+ for the synthesizer + (when unset)
149
+ ``config.producer.timeout_seconds`` resolution.
150
+ logger: Optional :class:`~syncade.logging.Logger`. Default
151
+ constructs a normal-verbosity logger.
152
+ adapter_factory: Reviewer adapter factory (default: the
153
+ production registry). Tests inject fakes.
154
+ synthesizer_adapter: Optional explicit synthesizer adapter
155
+ (default: resolved from the registry per round via ``config.synthesizer.provider``).
156
+ Tests pass :class:`FakeSynthesizerAdapter`. NOTE: when
157
+ a single adapter instance is supplied for a multi-round
158
+ run, it's reused across rounds — if the adapter has
159
+ per-call state (e.g. canned outputs that should vary
160
+ per round), the caller is responsible for that.
161
+ producer_adapter: Optional explicit producer adapter
162
+ (default: registry lookup using
163
+ ``config.producer.provider``). Tests pass
164
+ :class:`FakeProducerAdapter`.
165
+ worktree_base: Override for the per-run worktree base
166
+ directory. When ``None``, falls back to
167
+ ``config.worktree_base`` (which itself defaults to
168
+ :data:`DEFAULT_WORKTREE_BASE` = ``/tmp/syncade/``).
169
+ Tests use ``tmp_path`` to avoid cross-run collisions.
170
+ two_dot: When ``True``, diff the literal ``base..HEAD`` range instead
171
+ of from the branch point. The escape hatch for "show me everything
172
+ between these two commits"; it re-introduces phantom deletions for
173
+ a branch that is behind its base, which is why it is opt-in.
174
+ force_dirty: When ``True``, bypasses the loop-mode dirty-
175
+ tree refusal. Plumbed from the CLI's ``--force-dirty``
176
+ flag. Only relevant when ``max_rounds > 1`` AND
177
+ ``snapshot.dirty_state in {"tracked", "both"}``.
178
+ pr_doc_artifact_name: Fresh-run only. When set, copy the
179
+ input PR doc into ``<run_dir>/<name>`` before writing
180
+ ``run-init.json`` and use that persisted artifact as the
181
+ PR doc for every round. This keeps generated inputs such
182
+ as ``--openspec`` resumable after their staging tempfile
183
+ is removed.
184
+
185
+ Returns:
186
+ :class:`RunResult` carrying the loop's final exit code,
187
+ the per-round results, and the aggregate artifacts.
188
+
189
+ Raises:
190
+ FileNotFoundError / NotADirectoryError: For invalid inputs.
191
+ :class:`SnapshotError`: When ``repo_root`` isn't in a git
192
+ repo or the snapshot fails.
193
+ :class:`WorktreeError`: When a worktree can't be provisioned.
194
+ Bubbles up to the CLI which maps to exit 60.
195
+ """
196
+ logger = logger if logger is not None else Logger()
197
+
198
+ # Run-start timestamp captured once and threaded into every per-
199
+ # round manifest / summary so they all agree on when the RUN
200
+ # began (not when each file happened to be written).
201
+ started_at = datetime.now(tz=UTC)
202
+
203
+ # --- Input validation -------------------------------------------
204
+ validate_run_inputs(repo_root, pr_doc_path)
205
+
206
+ repo_root = repo_root.resolve()
207
+ pr_doc_path = pr_doc_path.resolve()
208
+ repo_root = discover_repo_root(repo_root)
209
+
210
+ # --- Round-0 snapshot (the starting point for the whole loop) ---
211
+ # On resume, use the immutable OID from the plan rather than the caller-supplied
212
+ # symbolic ref — the symbolic ref may have moved since the original run.
213
+ logger.event(f"snapshotting repo at {repo_root}")
214
+ _resumed_base = resume_plan.base_oid if resume_plan is not None else None
215
+ _snapshot_base_ref = _resumed_base if _resumed_base is not None else base_ref
216
+ # A resumed base_oid is ALREADY the effective diff base the original run
217
+ # resolved, so re-deriving a branch point from it would move the review
218
+ # target — and would convert a `--two-dot` run to three-dot on resume.
219
+ snapshot = take_snapshot(
220
+ repo_root,
221
+ base_ref=_snapshot_base_ref,
222
+ three_dot=not two_dot and _resumed_base is None,
223
+ )
224
+ branch = snapshot.branch or "(detached HEAD)"
225
+ logger.event(f"snapshot taken — {snapshot.commit_sha[:12]} on {branch}")
226
+
227
+ state = snapshot.dirty_state
228
+
229
+ run_preflight(
230
+ config=config,
231
+ repo_root=repo_root,
232
+ pr_doc_path=pr_doc_path,
233
+ snapshot=snapshot,
234
+ state=state,
235
+ branch=branch,
236
+ resume_plan=resume_plan,
237
+ logger=logger,
238
+ force_dirty=force_dirty,
239
+ allow_default_branch=allow_default_branch,
240
+ )
241
+ # CLI passes config.worktree_base explicitly; direct API callers that omit the kwarg
242
+ # still get the configured base (not always DEFAULT_WORKTREE_BASE).
243
+ effective_worktree_base = worktree_base if worktree_base is not None else config.worktree_base
244
+ runs_root = repo_root / ".syncade" / "runs"
245
+ if resume_plan is None:
246
+ # Auto-prune BEFORE this run's directory exists, so the run we are about to
247
+ # start cannot be a candidate for its own pruning. Fresh runs only: a resume
248
+ # reuses an existing run dir and may still need its transcripts, so pruning
249
+ # there is risk without benefit.
250
+ _autoprune_old_transcripts(
251
+ repo_root, logger, keep=config.gc.keep, max_age_days=config.gc.max_age_days
252
+ )
253
+
254
+ # FRESH RUN: generate run-id with collision resolution and
255
+ # mkdir the run directory.
256
+ base_run_id = generate_run_id()
257
+ run_id = base_run_id
258
+ attempt = 1
259
+ while True:
260
+ run_dir = runs_root / run_id
261
+ tmp_run_dir = effective_worktree_base / run_id
262
+ # Reserve the global /tmp worktree dir atomically rather than
263
+ # check-then-act on .exists(): mkdir(exist_ok=False) wins or
264
+ # raises, so a concurrent run racing on the same timestamp
265
+ # run-id cannot also claim this tmp subtree.
266
+ try:
267
+ tmp_run_dir.mkdir(parents=True, exist_ok=False)
268
+ except FileExistsError:
269
+ attempt += 1
270
+ run_id = f"{base_run_id}-{attempt}"
271
+ if attempt > 100:
272
+ raise RuntimeError(
273
+ f"could not find a free run_id after {attempt} attempts "
274
+ f"(base={base_run_id!r}); both <repo>/.syncade/runs/ "
275
+ f"and {effective_worktree_base}/ have stale entries — clean one up"
276
+ ) from None
277
+ continue
278
+ except OSError as exc:
279
+ raise WorktreeError(
280
+ f"cannot create run directory under worktree_base "
281
+ f"{effective_worktree_base!r}: {exc}"
282
+ ) from exc
283
+ try:
284
+ run_dir.mkdir(parents=True, exist_ok=False)
285
+ break
286
+ except FileExistsError:
287
+ # Release the tmp reservation we just took for this id so a
288
+ # run-dir collision does not strand an empty /tmp subtree.
289
+ tmp_run_dir.rmdir()
290
+ attempt += 1
291
+ run_id = f"{base_run_id}-{attempt}"
292
+ if attempt > 100:
293
+ raise
294
+ _ensure_runs_gitignore(runs_root)
295
+ # Begin the breadcrumb immediately after the run dir exists. The
296
+ # post-begin setup (PR-doc artifact copy, persist_run_init) is guarded
297
+ # below so any signal or exception in that window finalizes status.json
298
+ # rather than leaving it stuck as "running".
299
+ run_status.begin(run_dir, started_at)
300
+
301
+ try:
302
+ if pr_doc_artifact_name is not None:
303
+ artifact_name = Path(pr_doc_artifact_name).name
304
+ if not artifact_name:
305
+ raise ValueError("pr_doc_artifact_name must include a filename")
306
+ persisted_pr_doc_path = (run_dir / artifact_name).resolve()
307
+ atomic_write_text(
308
+ persisted_pr_doc_path,
309
+ pr_doc_path.read_text(encoding="utf-8"),
310
+ )
311
+ pr_doc_path = persisted_pr_doc_path
312
+
313
+ # Capture the run's initial state before any round directory is
314
+ # created. Resume reads this artifact but does not rewrite it.
315
+ persist_run_init(
316
+ run_dir,
317
+ syncade_version=__version__,
318
+ started_at=started_at,
319
+ pr_doc_path=pr_doc_path,
320
+ base_ref=base_ref,
321
+ base_oid=snapshot.base_oid,
322
+ starting_sha=snapshot.commit_sha,
323
+ operator_branch=snapshot.branch,
324
+ max_rounds=config.loop.max_rounds,
325
+ config=config,
326
+ )
327
+ except KeyboardInterrupt:
328
+ if not run_status.received_signal():
329
+ run_status.finalize_active("exception:KeyboardInterrupt", None)
330
+ raise
331
+ except BaseException as _exc:
332
+ run_status.finalize_active(f"exception:{type(_exc).__name__}", None)
333
+ raise
334
+ else:
335
+ # Resume reuses the original run-id and run directory and leaves
336
+ # run-init.json unchanged.
337
+ run_id = resume_plan.run_id
338
+ run_dir = resume_plan.run_dir
339
+ _ensure_runs_gitignore(runs_root)
340
+ # Register the breadcrumb immediately for resume: the reused run dir already
341
+ # holds a (possibly stale `running`) status.json from the original run, and
342
+ # the drift-check / rehydration below can raise or be signalled. Beginning
343
+ # here closes the pre-begin gap where a signal would leave the OLD status.json
344
+ # stale while stderr claims the signal was recorded.
345
+ run_status.begin(run_dir, started_at)
346
+
347
+ try:
348
+ # Tree-drift check: refuse unless the operator's current HEAD
349
+ # matches the resumed round's expected SHA. --force-drift
350
+ # bypasses the check (the resumed round then snapshots from
351
+ # current HEAD; see the "Resumed under tree drift" log line
352
+ # in the resumed round's summary.md).
353
+ if not force_drift:
354
+ try:
355
+ check_tree_drift(
356
+ repo_root,
357
+ expected_sha=resume_plan.expected_sha,
358
+ expected_branch=resume_plan.expected_branch,
359
+ )
360
+ except Exception as exc:
361
+ # TreeDriftError + ResumeError both bubble up here.
362
+ # The CLI maps both to exit 60. Reraise without
363
+ # additional context.
364
+ raise WorktreeError(str(exc)) from exc
365
+
366
+ # Drop the resumed round's stale state if any. The on-disk
367
+ # directory and the worktree subtree at /tmp/syncade/<run-id>/
368
+ # round-N/ are remnants of the aborted attempt; we re-run from
369
+ # snapshot. Both removals go through _safe_resume_rmtree, which
370
+ # applies the same containment + identity guards GC uses so a
371
+ # swapped symlink or out-of-base path can never redirect the
372
+ # delete; missing targets safely no-op (idempotent). Only the
373
+ # EXTERNAL worktree subtree is reaped (reap=True), matching GC's
374
+ # worktree-tree removal; the persisted .syncade/runs artifact dir
375
+ # uses a plain guarded rmtree (reap=False) — an operator may be
376
+ # inspecting it, and SIGKILL would over-reach inside .syncade/runs/ (M2).
377
+ #
378
+ # Exception: for a budget-abort-before-producer resume the round
379
+ # directory holds a complete review bundle that we will rehydrate —
380
+ # do NOT delete it. No external worktree was created (the producer
381
+ # never ran), so there is nothing to prune.
382
+ _is_budget_abort_resume = resume_plan.budget_aborted_before_producer_round is not None
383
+ if not _is_budget_abort_resume:
384
+ resumed_round_dir = run_dir / f"round-{resume_plan.resumed_round}"
385
+ _safe_resume_rmtree(resumed_round_dir, runs_root, repo_root, reap=False)
386
+ resumed_worktree_dir = (
387
+ effective_worktree_base / run_id / f"round-{resume_plan.resumed_round}"
388
+ )
389
+ _safe_resume_rmtree(
390
+ resumed_worktree_dir, effective_worktree_base, repo_root, reap=True
391
+ )
392
+ # Terminal rounds that preserve worktrees also leave git worktree
393
+ # registry entries. The rmtree above removed the dirs, but stale registry
394
+ from syncade.process import run_subprocess as _run_subprocess
395
+
396
+ try:
397
+ _run_subprocess(["git", "worktree", "prune"], cwd=repo_root, timeout=10.0)
398
+ except Exception:
399
+ pass
400
+
401
+ # Warn on syncade-version drift. This is informational only.
402
+ if resume_plan.syncade_version and resume_plan.syncade_version != __version__:
403
+ logger.warning(
404
+ f"orchestrator: resuming run {run_id} started under "
405
+ f"syncade {resume_plan.syncade_version}; current "
406
+ f"syncade is {__version__}. Behavior may differ; "
407
+ f"inspect the resumed round's artifacts if surprised."
408
+ )
409
+
410
+ # When force-drift is used, surface the drift now and in the resumed
411
+ # round's durable summary.md annotation.
412
+ if force_drift:
413
+ logger.warning(
414
+ f"orchestrator: resuming run {run_id} under tree drift "
415
+ f"(--force-drift). The resumed round-{resume_plan.resumed_round} "
416
+ f"will snapshot from current HEAD; the original run's "
417
+ f"expected SHA was {resume_plan.expected_sha[:12]}. "
418
+ f"Cross-round context from prior rounds references findings "
419
+ f"against the ORIGINAL tree state, not the current one — "
420
+ f"reviewers may surface inconsistencies."
421
+ )
422
+ except KeyboardInterrupt:
423
+ if not run_status.received_signal():
424
+ run_status.finalize_active("exception:KeyboardInterrupt", None)
425
+ raise
426
+ except WorktreeError as _exc:
427
+ run_status.finalize_active(f"exception:{type(_exc).__name__}", WORKTREE_ERROR)
428
+ raise
429
+ except BaseException as _exc:
430
+ run_status.finalize_active(f"exception:{type(_exc).__name__}", None)
431
+ raise
432
+
433
+ try:
434
+ # --- Timeout resolution -----------------------------------------
435
+ resolved_timeout = (
436
+ timeout_seconds if timeout_seconds is not None else config.loop.timeout_seconds
437
+ )
438
+ # Producer timeout: explicit > shared reviewer timeout (per
439
+ # config.producer.timeout_seconds docs).
440
+ resolved_producer_timeout = (
441
+ config.producer.timeout_seconds
442
+ if config.producer.timeout_seconds is not None
443
+ else resolved_timeout
444
+ )
445
+
446
+ # --- Loop state -------------------------------------------------
447
+ round_results: list[RoundResult] = []
448
+ round_artifacts_list: list[RoundArtifacts] = []
449
+ # Track every WorktreeManager opened during the loop. They defer cleanup so
450
+ # finalization can preserve diagnostic worktrees on operator-action exits
451
+ # and clean them on success/error exits.
452
+ managers_to_cleanup: list[WorktreeManager] = []
453
+ termination_reason: TerminationReason | None = None
454
+ final_exit_code: int = SUCCESS
455
+ branch_advanced_during_run = False
456
+ # PR-v2-11: running per-actor usage tally for the budget checks. Starts empty even on
457
+ # --resume (a fresh tally: the prior process already spent that; this run bounds only
458
+ # what IT spends), so a resumed loop is never aborted for a predecessor's cost.
459
+ run_usages: list[Usage] = []
460
+ budget_warned = False # the 80% heads-up fires once per run, not once per round
461
+ # Which ceiling tripped ("budget_tokens" | "budget_usd"), so the Budget section can
462
+ # name it (both set → first-to-trip, tokens checked first). None unless a budget abort.
463
+ budget_ceiling: str | None = None
464
+
465
+ # The current snapshot — refreshed each round when the previous
466
+ # round's producer advanced the branch.
467
+ current_snapshot = snapshot
468
+
469
+ # --- Resume rehydration -----------------------------------------
470
+ # Rehydrate each completed round and derive the resumed loop bounds.
471
+ resumed_round_start: int = 0
472
+ if resume_plan is not None:
473
+ (
474
+ round_results,
475
+ round_artifacts_list,
476
+ resumed_round_start,
477
+ branch_advanced_during_run,
478
+ config,
479
+ ) = _rehydrate_resume_state(resume_plan, run_dir, run_id, config)
480
+
481
+ # --- The round loop ---------------------------------------------
482
+ # Each iteration mutates the round/result/cleanup lists in place and
483
+ # returns a continue/break signal.
484
+ for round_idx in range(resumed_round_start, config.loop.max_rounds):
485
+ # PR-v2-11: refuse to START a round once prior rounds' accumulated spend already
486
+ # crossed a budget (round 0 always runs — the tally is empty). A no-op when no
487
+ # ceiling is active (the 0 sentinel), so an opted-out run's control flow is unchanged.
488
+ budget_ceiling = over_budget(run_usages, config.loop)
489
+ if budget_ceiling is not None:
490
+ final_exit_code = BUDGET_EXCEEDED
491
+ termination_reason = "budget_exceeded"
492
+ break
493
+ # Advisory heads-up at the SAME boundary the ceiling is enforced at, so the
494
+ # operator sees the stop coming with a round left to react in (PR-h-field-06).
495
+ # Checked only when over_budget stayed silent, so the two never both speak. Fired
496
+ # ONCE per run: a warning repeated every round is a warning nobody reads, and it
497
+ # would be loudest exactly when the operator is already watching a long run.
498
+ if not budget_warned:
499
+ approaching = approaching_budget(run_usages, config.loop)
500
+ if approaching is not None:
501
+ budget_warned = True
502
+ logger.warning(_approaching_budget_line(approaching, run_usages, config.loop))
503
+ step = _run_round_step(
504
+ round_idx=round_idx,
505
+ current_snapshot=current_snapshot,
506
+ resumed_round_start=resumed_round_start,
507
+ repo_root=repo_root,
508
+ pr_doc_path=pr_doc_path,
509
+ run_id=run_id,
510
+ run_dir=run_dir,
511
+ config=config,
512
+ base_ref=base_ref,
513
+ resolved_timeout=resolved_timeout,
514
+ resolved_producer_timeout=resolved_producer_timeout,
515
+ adapter_factory=adapter_factory,
516
+ synthesizer_adapter=synthesizer_adapter,
517
+ producer_adapter=producer_adapter,
518
+ effective_worktree_base=effective_worktree_base,
519
+ logger=logger,
520
+ started_at=started_at,
521
+ managers_to_cleanup=managers_to_cleanup,
522
+ round_results=round_results,
523
+ round_artifacts_list=round_artifacts_list,
524
+ resume_plan=resume_plan,
525
+ operator_decision=operator_decision,
526
+ force_drift=force_drift,
527
+ prior_usages=run_usages,
528
+ branch_advanced_during_run=branch_advanced_during_run,
529
+ budget_warned=budget_warned,
530
+ )
531
+ current_snapshot = step.current_snapshot
532
+ if step.branch_advanced:
533
+ branch_advanced_during_run = True
534
+ if step.budget_warned:
535
+ budget_warned = True
536
+ # Accumulate the round just run (reviewers + judge + any producer) BEFORE the
537
+ # break check, so the final tally the summary/metrics report includes the round
538
+ # that crossed — whether it broke here or the step's pre-producer check broke it.
539
+ # For a budget-abort-before-producer resume the review bundle was paid for in the
540
+ # prior process; only the producer cost counts against the fresh tally.
541
+ _is_budget_abort_round = (
542
+ resume_plan is not None
543
+ and resume_plan.budget_aborted_before_producer_round == round_idx
544
+ )
545
+ run_usages.extend(
546
+ producer_only_usages(round_results[-1])
547
+ if _is_budget_abort_round
548
+ else round_usages(round_results[-1])
549
+ )
550
+ if step.action == "break":
551
+ final_exit_code = step.final_exit_code
552
+ termination_reason = step.termination_reason
553
+ budget_ceiling = step.budget_ceiling # None unless a pre-producer budget abort
554
+ break
555
+
556
+ # --- Loop terminated --------------------------------------------
557
+ # completed_at captured here (in the rebound body) so the patched
558
+ # ``datetime`` lookup stays inside run_review; passed into
559
+ # _finalize_run, which therefore references no monkeypatched name.
560
+ completed_at = datetime.now(tz=UTC)
561
+ result = _finalize_run(
562
+ run_dir=run_dir,
563
+ run_id=run_id,
564
+ repo_root=repo_root,
565
+ snapshot=snapshot,
566
+ config=config,
567
+ pr_doc_path=pr_doc_path,
568
+ round_results=round_results,
569
+ round_artifacts_list=round_artifacts_list,
570
+ final_exit_code=final_exit_code,
571
+ termination_reason=termination_reason,
572
+ started_at=started_at,
573
+ completed_at=completed_at,
574
+ branch_advanced_during_run=branch_advanced_during_run,
575
+ managers_to_cleanup=managers_to_cleanup,
576
+ effective_worktree_base=effective_worktree_base,
577
+ logger=logger,
578
+ run_usages=run_usages,
579
+ budget_ceiling=budget_ceiling,
580
+ )
581
+ # Normal exit: _finalize_run already finalized status on the
582
+ # test_worktree_error re-raise path (mechanical reason recorded
583
+ # before raise). For every other normal return, finalize here.
584
+ run_status.finalize_active(result.termination_reason or "unknown", result.exit_code)
585
+ return result
586
+ except KeyboardInterrupt:
587
+ # Signal-induced KI: run_status._handler recorded the signum and
588
+ # raised KI. Preserve the active breadcrumb so the CLI's
589
+ # finalize_signal() can write signal:<NAME> + 128+signum. For a
590
+ # pure (non-signal) KI there is no CLI handler waiting — finalize here.
591
+ if not run_status.received_signal():
592
+ run_status.finalize_active("exception:KeyboardInterrupt", None)
593
+ raise
594
+ except (WorktreeError, SnapshotError, ResumeError) as exc:
595
+ # These all map to exit 60 in the CLI. Finalize with that code so status.json
596
+ # matches the process exit for both direct API callers and CLI callers (whose
597
+ # typed handler is a no-op after this clears _active) — not a null exit_code
598
+ # that would disagree with the mapped exit.
599
+ run_status.finalize_active(f"exception:{type(exc).__name__}", WORKTREE_ERROR)
600
+ raise
601
+ except Exception as exc:
602
+ # Unexpected exception: finalize for direct API callers (no CLI
603
+ # wrapper). If _finalize_run already finalized (mechanical re-raise
604
+ # path), _active is None and this is a no-op.
605
+ run_status.finalize_active(f"exception:{type(exc).__name__}", None)
606
+ raise
607
+ except BaseException as exc:
608
+ # A parent-side SystemExit (or any non-Exception BaseException that is not the
609
+ # KeyboardInterrupt handled above) must not leave the breadcrumb `running`.
610
+ run_status.finalize_active(f"exception:{type(exc).__name__}", None)
611
+ raise