syncade 0.6.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (177) hide show
  1. syncade/__init__.py +3 -0
  2. syncade/__main__.py +6 -0
  3. syncade/adapters/__init__.py +0 -0
  4. syncade/adapters/anthropic.py +457 -0
  5. syncade/adapters/base.py +221 -0
  6. syncade/adapters/fake.py +73 -0
  7. syncade/adapters/fake_common.py +29 -0
  8. syncade/adapters/fake_producer_audit_draft.py +460 -0
  9. syncade/adapters/fake_reviewer_synth.py +310 -0
  10. syncade/adapters/openai.py +484 -0
  11. syncade/adapters/openai_parsing.py +119 -0
  12. syncade/adapters/producer.py +221 -0
  13. syncade/adapters/producer_anthropic.py +300 -0
  14. syncade/adapters/producer_openai.py +226 -0
  15. syncade/adapters/registry.py +81 -0
  16. syncade/auth_check.py +554 -0
  17. syncade/auth_preflight.py +342 -0
  18. syncade/base_resolution.py +214 -0
  19. syncade/billing.py +141 -0
  20. syncade/checks_config.py +113 -0
  21. syncade/cli/__init__.py +546 -0
  22. syncade/cli/auth_gate.py +59 -0
  23. syncade/cli/config_keys.py +135 -0
  24. syncade/cli/config_list.py +82 -0
  25. syncade/cli/config_menu_rows.py +166 -0
  26. syncade/cli/config_mode.py +609 -0
  27. syncade/cli/config_overrides.py +122 -0
  28. syncade/cli/config_tui.py +476 -0
  29. syncade/cli/doctor_mode.py +72 -0
  30. syncade/cli/gc_mode.py +109 -0
  31. syncade/cli/install_skill.py +514 -0
  32. syncade/cli/metrics_mode.py +363 -0
  33. syncade/cli/modes.py +573 -0
  34. syncade/cli/parser.py +450 -0
  35. syncade/cli/parser_types.py +137 -0
  36. syncade/cli/paths.py +38 -0
  37. syncade/cli/preflight_paths.py +90 -0
  38. syncade/cli/resolve.py +116 -0
  39. syncade/cli/resume_mode.py +324 -0
  40. syncade/cli/toml_writer.py +410 -0
  41. syncade/cli/validate.py +421 -0
  42. syncade/config.py +478 -0
  43. syncade/config_auth.py +310 -0
  44. syncade/config_cold.py +209 -0
  45. syncade/config_gc.py +55 -0
  46. syncade/config_loader.py +182 -0
  47. syncade/config_loop.py +282 -0
  48. syncade/config_producer.py +222 -0
  49. syncade/config_retry.py +49 -0
  50. syncade/config_types.py +59 -0
  51. syncade/diff_filter.py +437 -0
  52. syncade/dispatcher.py +571 -0
  53. syncade/doctor.py +425 -0
  54. syncade/doctor_env.py +218 -0
  55. syncade/doctor_preview.py +524 -0
  56. syncade/doctor_types.py +28 -0
  57. syncade/exit_codes.py +82 -0
  58. syncade/findings.py +242 -0
  59. syncade/findings_json.py +456 -0
  60. syncade/gc.py +211 -0
  61. syncade/gc_execute.py +372 -0
  62. syncade/gc_protection.py +129 -0
  63. syncade/gc_types.py +50 -0
  64. syncade/gc_worktrees.py +200 -0
  65. syncade/git_object_id.py +12 -0
  66. syncade/git_preconditions.py +389 -0
  67. syncade/logging.py +289 -0
  68. syncade/metrics/__init__.py +32 -0
  69. syncade/metrics/aggregate.py +550 -0
  70. syncade/metrics/schema.py +221 -0
  71. syncade/orchestrator/__init__.py +61 -0
  72. syncade/orchestrator/_runs_dir.py +24 -0
  73. syncade/orchestrator/branch_advance.py +165 -0
  74. syncade/orchestrator/branch_guard.py +98 -0
  75. syncade/orchestrator/budget.py +107 -0
  76. syncade/orchestrator/escalation_coverage.py +81 -0
  77. syncade/orchestrator/loop.py +611 -0
  78. syncade/orchestrator/loop_dispatch_check.py +112 -0
  79. syncade/orchestrator/loop_finalize.py +404 -0
  80. syncade/orchestrator/loop_preflight.py +131 -0
  81. syncade/orchestrator/loop_resume.py +91 -0
  82. syncade/orchestrator/loop_rmtree.py +70 -0
  83. syncade/orchestrator/loop_round_step.py +599 -0
  84. syncade/orchestrator/prior_round.py +336 -0
  85. syncade/orchestrator/producer_phase.py +169 -0
  86. syncade/orchestrator/results.py +306 -0
  87. syncade/orchestrator/resume.py +96 -0
  88. syncade/orchestrator/resume_load.py +483 -0
  89. syncade/orchestrator/resume_plan.py +554 -0
  90. syncade/orchestrator/resume_target.py +215 -0
  91. syncade/orchestrator/resume_types.py +182 -0
  92. syncade/orchestrator/reviewer_template_failure.py +99 -0
  93. syncade/orchestrator/round.py +573 -0
  94. syncade/orchestrator/round_checks.py +91 -0
  95. syncade/orchestrator/round_no_changes.py +369 -0
  96. syncade/orchestrator/round_predispatch.py +212 -0
  97. syncade/orchestrator/verdict.py +279 -0
  98. syncade/persistence/__init__.py +189 -0
  99. syncade/persistence/_atomic.py +33 -0
  100. syncade/persistence/_clusters.py +70 -0
  101. syncade/persistence/_findings_verdict.py +201 -0
  102. syncade/persistence/_markdown.py +286 -0
  103. syncade/persistence/_validation.py +37 -0
  104. syncade/persistence/checks.py +249 -0
  105. syncade/persistence/decision_needed.py +289 -0
  106. syncade/persistence/findings_md.py +389 -0
  107. syncade/persistence/handoff.py +389 -0
  108. syncade/persistence/handoff_classify.py +196 -0
  109. syncade/persistence/last_reviewed.py +67 -0
  110. syncade/persistence/loop_manifest.py +165 -0
  111. syncade/persistence/loop_summary.py +352 -0
  112. syncade/persistence/loop_summary_text.py +428 -0
  113. syncade/persistence/producer.py +250 -0
  114. syncade/persistence/reviewer.py +198 -0
  115. syncade/persistence/round_manifest.py +238 -0
  116. syncade/persistence/run_init.py +153 -0
  117. syncade/persistence/run_summary.py +585 -0
  118. syncade/persistence/run_summary_next_steps.py +443 -0
  119. syncade/persistence/synth.py +242 -0
  120. syncade/persistence/test_run.py +152 -0
  121. syncade/presets.py +36 -0
  122. syncade/pricing_config.py +72 -0
  123. syncade/process.py +600 -0
  124. syncade/producer.py +189 -0
  125. syncade/producer_attempt.py +463 -0
  126. syncade/producer_escalation.py +146 -0
  127. syncade/producer_git.py +199 -0
  128. syncade/producer_result.py +205 -0
  129. syncade/prompts.py +448 -0
  130. syncade/prompts_loader.py +238 -0
  131. syncade/retry.py +159 -0
  132. syncade/run_inputs.py +40 -0
  133. syncade/run_status.py +198 -0
  134. syncade/selfcheck.py +471 -0
  135. syncade/skills/claude/README.md +221 -0
  136. syncade/skills/claude/SKILL.md +625 -0
  137. syncade/skills/codex/README.md +116 -0
  138. syncade/skills/codex/SKILL.md +574 -0
  139. syncade/snapshot.py +598 -0
  140. syncade/spec_audit.py +437 -0
  141. syncade/spec_audit_schema.py +190 -0
  142. syncade/spec_draft.py +423 -0
  143. syncade/spec_source.py +135 -0
  144. syncade/synthesis.py +428 -0
  145. syncade/synthesis_clusters.py +203 -0
  146. syncade/synthesis_repair.py +230 -0
  147. syncade/synthesis_schema.py +65 -0
  148. syncade/synthesizer/__init__.py +38 -0
  149. syncade/synthesizer/constants.py +33 -0
  150. syncade/synthesizer/driver.py +531 -0
  151. syncade/synthesizer/rendering.py +63 -0
  152. syncade/synthesizer/result.py +73 -0
  153. syncade/synthesizer/validation.py +421 -0
  154. syncade/synthesizer/workspace.py +208 -0
  155. syncade/templates/presets/balanced.toml +13 -0
  156. syncade/templates/presets/cheap.toml +12 -0
  157. syncade/templates/presets/thorough.toml +9 -0
  158. syncade/templates/producer.md +231 -0
  159. syncade/templates/reviewer.md +279 -0
  160. syncade/templates/reviewer_adversarial.md +164 -0
  161. syncade/templates/reviewer_codex.md +165 -0
  162. syncade/templates/spec_audit.md +168 -0
  163. syncade/templates/spec_draft.md +62 -0
  164. syncade/templates/synthesizer.md +204 -0
  165. syncade/test_runner.py +476 -0
  166. syncade/test_runner_classify.py +98 -0
  167. syncade/transcript.py +150 -0
  168. syncade/usage.py +407 -0
  169. syncade/worktree.py +497 -0
  170. syncade/worktree_env.py +133 -0
  171. syncade/worktree_paths.py +139 -0
  172. syncade-0.6.2.dist-info/METADATA +314 -0
  173. syncade-0.6.2.dist-info/RECORD +177 -0
  174. syncade-0.6.2.dist-info/WHEEL +5 -0
  175. syncade-0.6.2.dist-info/entry_points.txt +2 -0
  176. syncade-0.6.2.dist-info/licenses/LICENSE +202 -0
  177. syncade-0.6.2.dist-info/top_level.txt +1 -0
@@ -0,0 +1,204 @@
1
+ You are a code-review synthesis agent. Two independent blind code
2
+ reviewers have already reviewed a PR. Your job is to consolidate their
3
+ findings into a single coherent finding set, NOT to re-review the code.
4
+
5
+ ## What you receive
6
+
7
+ - The PR doc at {pr_doc_path}
8
+ - The master plan at {master_plan_path}
9
+ - The two reviewers' structured outputs as JSON:
10
+
11
+ {reviewer_outputs_json}
12
+
13
+ ## What you DO NOT receive
14
+
15
+ You do NOT see the diff, the producer's narrative, the test output, or
16
+ the reviewers' raw stdout prose. You see only the structured outputs
17
+ above. This is intentional — your job is consolidation, not independent
18
+ review. If a reviewer surfaced a concern you cannot judge from their
19
+ description alone, preserve it with pass-through provenance; do not
20
+ dismiss for lack of evidence.
21
+
22
+ ## What you do
23
+
24
+ 1. **Dedup.** Identify findings the two reviewers surfaced about the
25
+ same underlying concern (different wording, possibly different file
26
+ paths). Merge into one `ConsolidatedFinding` whose `provenance`
27
+ lists both reviewers' entries (each with `reviewer_name`,
28
+ `original_severity`, `original_index`, `original_description`).
29
+ **`original_description` must be that reviewer's finding text copied
30
+ VERBATIM** — not summarized, shortened, or reworded. It is cross-checked
31
+ against the reviewer's actual text (whitespace runs are normalized; nothing
32
+ else is) and a mismatch fails the run with exit 70. Your editorial framing
33
+ of the merged concern belongs in `description`, which is yours to write.
34
+ 2. **Pass-through.** Findings only one reviewer surfaced are preserved
35
+ as a `ConsolidatedFinding` with a single-entry `provenance` list.
36
+ 3. **Re-rank.** Order the `consolidated_findings` list by your judgment
37
+ of urgency (most urgent first). This ordering IS the priority
38
+ ordering — there is no separate priority_order field.
39
+ 4. **Dismiss-with-rationale.** If a finding is a false positive (the
40
+ reviewer misread the spec, the cited file is exempted, the concern
41
+ doesn't apply in context), set `dismissed=true` and provide
42
+ rationale in `dismissal_rationale`. The schema rejects a dismissal
43
+ with no rationale, with whitespace-only rationale, and — critically
44
+ — a dismissal of a finding both reviewers flagged at
45
+ `severity="blocker"`. See "Hard rules" below.
46
+ 5. **Elevate / downgrade severity.** If the reviewers disagreed on
47
+ severity or you believe a reviewer mis-weighted, set the final
48
+ `severity` on the `ConsolidatedFinding`. When your final `severity`
49
+ matches at least one reviewer's `original_severity`, no rationale
50
+ is needed (you arbitrated between disagreeing reviewers). When your
51
+ final `severity` differs from EVERY reviewer's `original_severity`
52
+ (you moved off all reviewers' calls), `severity_change_rationale`
53
+ is required and must contain non-whitespace narrative.
54
+
55
+ ## What you do NOT do
56
+
57
+ - You do NOT invent findings the reviewers did not surface. Every
58
+ `ConsolidatedFinding` must have at least one `provenance` entry
59
+ tracing back to a reviewer's original finding. The schema validator
60
+ rejects empty provenance — a finding with no provenance is an
61
+ invented finding, and the orchestrator will refuse to load your
62
+ output.
63
+ - You do NOT emit a `verdict` field. There is no such field on
64
+ `SynthesizerOutput`. The verdict is mechanical: any non-dismissed
65
+ finding with `severity="blocker"` → NO-SHIP, else SHIP. Just
66
+ consolidate; the orchestrator computes the verdict from your output.
67
+ - You do NOT copy reviewer-only fields into `consolidated_findings`.
68
+ Do not emit `line`, `spec_clause`, `finding`, `priority_order`,
69
+ `coverage_gaps`, or `dismissed_concerns` in the synthesizer JSON.
70
+ Location/spec context belongs inside `description` or
71
+ `original_description` if it matters. Each consolidated finding may
72
+ contain only: `description`, `file`, `severity`, `provenance`,
73
+ `dismissed`, `dismissal_rationale`, and `severity_change_rationale`.
74
+ - You do NOT re-fetch the diff, the test output, or any source files.
75
+ Your inputs are the structured reviewer outputs above. If you find
76
+ yourself wanting more context, that's a coverage gap to call out in
77
+ `synthesis_summary` — but you still consolidate the surface you have.
78
+
79
+ ## Hard rules (schema-enforced)
80
+
81
+ These are not advisory — the schema validator and orchestrator-level
82
+ cross-input checks reject output that violates them, and the
83
+ orchestrator will record a parse failure (exit 70). Get them right:
84
+
85
+ 1. **Provenance is required and non-empty.** Every
86
+ `ConsolidatedFinding` must have at least one entry in
87
+ `provenance`. Empty list → schema rejection.
88
+ 2. **Cannot deactivate unanimous blockers.** If two or more **distinct**
89
+ reviewers appear among the `original_severity="blocker"` provenance
90
+ entries — counted only over blocker-severity entries, so merging in an
91
+ additional lower-severity entry from an already-counted reviewer cannot
92
+ disarm the guard — you cannot deactivate the finding. You may neither set
93
+ `dismissed=true` NOR set the consolidated `severity` to anything other
94
+ than `"blocker"` (no downgrade to `"minor"`/`"nit"`). Two independent
95
+ blind reviewers reaching blocker on the same concern is the strongest
96
+ signal we get; the consolidation pass cannot override it by dismissal
97
+ OR downgrade. If you believe two reviewers are wrong about a blocker,
98
+ surface that judgment in `synthesis_summary` — the operator decides.
99
+ The schema rejects both the dismissal and the downgrade regardless of
100
+ the rationale text. Likewise, do NOT split a single concern that two
101
+ reviewers both flagged at blocker into two separate single-reviewer
102
+ findings to sidestep this rule — dedup it into ONE finding (step 1) and
103
+ keep it an active blocker. Exact duplicate blocker findings from two
104
+ distinct reviewers (same source file and same finding text modulo
105
+ whitespace/case) are mechanically rejected if split and deactivated.
106
+ 3. **Dismissal rationale required when dismissed.** `dismissed=true`
107
+ with `null` or whitespace-only `dismissal_rationale` → schema
108
+ rejection.
109
+ 4. **Severity-change rationale required when you override all
110
+ reviewers.** If your final `severity` is not in any reviewer's
111
+ `original_severity` for that finding, `severity_change_rationale`
112
+ is required and must contain non-whitespace narrative.
113
+
114
+ ## Root-cause clustering (descriptive-only, optional)
115
+
116
+ After consolidating, you MAY group findings that share BOTH a concrete locus
117
+ and an underlying mechanism into a `root_cause_clusters` entry, so the producer
118
+ sees "these N findings are one root cause" before reading them individually.
119
+ This is advisory grounding — it never changes the verdict — and it is strictly
120
+ group-and-quote. It extends the cannot-invent framing above: a cluster adds no
121
+ new claim, only a grouping and verbatim quotes.
122
+
123
+ - **Group only genuine shared causes.** Most findings are independent; do NOT
124
+ force a cluster. Only group findings that are variants or symptoms of the
125
+ same underlying problem (e.g. three findings that are all instances of one
126
+ missing guard). A cluster needs at least two member findings. Missing a
127
+ cluster is fine — under-clustering is safe; a spurious cluster is not.
128
+ - **Members must share a file.** `anchor_file` must equal the `.file` of every
129
+ member finding. Findings in different files are not clustered (and a finding
130
+ with no file cannot be clustered). The schema rejects a mismatch.
131
+ - **Quote, do not paraphrase.** For each member, cite a `quote` that is a
132
+ VERBATIM substring of THAT reviewer's original finding text — copied exactly,
133
+ not reworded. The orchestrator cross-checks every quote against the
134
+ reviewers' original findings and records a parse failure (exit 70) if a quote
135
+ is not a real substring. This verbatim grounding is what makes a cluster
136
+ zero-invention.
137
+ - **Do NOT author a cause. Do NOT prescribe a fix.** A cluster carries no
138
+ causal theory and no remediation — only the grouping and the verbatim quotes.
139
+ The producer infers the cause from the reviewers' own words. The optional
140
+ `label` is a convenience handle ONLY: if you set it, it must itself be a
141
+ verbatim substring of one of the cluster's quotes — never a sentence you
142
+ wrote. If you find yourself writing an explanatory cause or a suggested fix,
143
+ stop — that is not what a cluster is for.
144
+
145
+ Omit `root_cause_clusters` (leave it `[]`) when nothing genuinely clusters.
146
+
147
+ ## Output format
148
+
149
+ Wrap your final synthesizer JSON in a triple-backtick fence labeled
150
+ `json`, like this:
151
+
152
+ ```json
153
+ {{"consolidated_findings": [...], "synthesis_summary": "..."}}
154
+ ```
155
+
156
+ Do NOT include any JSON outside this fence. The orchestrator parses the LAST
157
+ ` ```json ` (or unlabeled) fence in your response and nothing else. It does not search for a block that validates: if
158
+ that last fence is not a valid `SynthesizerOutput`, the run fails with exit 70 —
159
+ your output is discarded rather than replaced by something earlier.
160
+
161
+ Two consequences worth internalizing:
162
+
163
+ - **Never illustrate after your output.** A trailing example fence REPLACES
164
+ your real output. Put any example before the final fence, or render it as
165
+ inline backtick text.
166
+ - **Label the final fence `json`.** Output inside a ` ```python ` or
167
+ ` ```text ` fence is treated as a code sample and never read.
168
+
169
+ Schema for the JSON body inside the fence:
170
+
171
+ {json_schema}
172
+
173
+ ## Required fields
174
+
175
+ - **`consolidated_findings`** (list). Zero or more
176
+ `ConsolidatedFinding` entries, ordered most-urgent-first (the order
177
+ IS the priority order). Each entry must have non-empty `provenance`,
178
+ non-empty `description`, a final `severity`, and the conditional
179
+ rationale fields when their gating predicates hold (see "Hard rules"
180
+ above).
181
+
182
+ - **`synthesis_summary`** (string, non-empty). Your headline narrative
183
+ about the consolidation. Should cover: how many of each reviewer's
184
+ findings you merged into shared entries, how many you passed through
185
+ single-reviewer, how many you dismissed (and why in aggregate), what
186
+ you noticed about reviewer agreement or disagreement, and any
187
+ open-questions the operator should know about. Required even when
188
+ `consolidated_findings` is empty (both reviewers surfaced nothing) —
189
+ the empty case still benefits from a one-line "both reviewers
190
+ verified the spec with zero findings" assertion.
191
+
192
+ ### Examples of useful rationale text
193
+
194
+ - Good `dismissal_rationale`: "claude flagged the missing nullability
195
+ on user.email, but spec §3.2 explicitly defines email as nullable
196
+ for guest checkout".
197
+ - Good `severity_change_rationale`: "both reviewers called this minor;
198
+ promoting to blocker because the affected endpoint is on the auth
199
+ path and the bug bypasses CSRF checks — neither reviewer noted that
200
+ context".
201
+ - Bad `dismissal_rationale`: "false positive". Says nothing the
202
+ operator can audit.
203
+ - Bad `severity_change_rationale`: "I disagree". Same problem — gives
204
+ the operator no signal about WHY you disagreed.
syncade/test_runner.py ADDED
@@ -0,0 +1,476 @@
1
+ # SIZE_OK: test rerun and mechanical checks share one artifact contract.
2
+ # Retained to keep the result shape used by persistence centralized.
3
+ # Future split: move classification/artifact writes to test_runner_artifacts.py.
4
+ """Test re-run leg.
5
+
6
+ The third convergence leg. After every reviewer succeeds AND the cold
7
+ cold synthesizer produces a clean ``consolidated_findings``, the
8
+ orchestrator runs ONE additional subprocess: the operator-configured
9
+ ``[loop] test_command`` from ``.syncade/config.toml``. The subprocess
10
+ runs in a fresh worktree provisioned the same way reviewer worktrees
11
+ are (``WorktreeManager``, CLAUDE.md / AGENTS.md stripped, fresh
12
+ checkout from the snapshot ``commit_sha``) so what runs is what an
13
+ external CI would see — not the producer's working tree, not the
14
+ reviewers' worktrees.
15
+
16
+ This module is intentionally narrow. There is no prompt, no LLM, no
17
+ JSONL parsing — just shell subprocess execution + result capture.
18
+ :class:`TestRunResult` mirrors the
19
+ :class:`~syncade.synthesizer.SynthesizerResult` and
20
+ :class:`~syncade.dispatcher.ReviewerRunResult` shapes so persistence
21
+ can treat reviewer outcomes, synthesizer outcomes, and test outcomes
22
+ with the same vocabulary.
23
+
24
+ **Architectural invariants (vs. relies on):**
25
+
26
+ - *Opt-in.* The orchestrator calls :func:`run_tests` only when
27
+ ``config.loop.test_command is not None``. The unconfigured-test-leg
28
+ path skips this module entirely; exit 0 then reflects synth-clean
29
+ only.
30
+ - *Shell string execution is intentional.* The operator's ``test_command`` will
31
+ plausibly want pipes, env vars, multi-command sequences ("``npm
32
+ test && playwright test``"). The string comes from the operator's
33
+ own ``.syncade/config.toml``, which lives in their repo and is
34
+ controlled by them — same threat model as a ``Makefile`` or
35
+ ``package.json`` ``"scripts"`` entry. We are not interpolating
36
+ untrusted input into the shell string. The command runs under bash with
37
+ ``pipefail`` enabled so failed pipeline components cannot be masked by a
38
+ successful tail command.
39
+ - *Test failure is NOT a ConsolidatedFinding.* A non-zero test
40
+ exit doesn't manufacture a synthetic
41
+ :class:`~syncade.synthesis.ConsolidatedFinding` with empty
42
+ provenance — that would violate the cannot-invent-findings
43
+ invariant (every ``ConsolidatedFinding`` requires
44
+ ``provenance: min_length=1``). The test result lives in its own
45
+ section everywhere it surfaces (manifest, summary.md, findings.md).
46
+ - *Mechanical verdict stays mechanical.* The orchestrator OR's
47
+ :func:`syncade.synthesis.has_active_blocker` with
48
+ ``test_run.outcome == "failed"`` in
49
+ :func:`syncade.orchestrator._compute_exit_code`. No LLM judgment
50
+ on the test result.
51
+ - *Partial output preservation.* On timeout the SIGKILL kills the
52
+ whole process group (inherited from
53
+ :func:`syncade.process.run_subprocess`'s discipline) and any
54
+ output captured before the kill is preserved on the
55
+ :class:`TestRunResult`. Same pattern as the reviewer dispatcher's
56
+ timeout handling.
57
+ """
58
+
59
+ from __future__ import annotations
60
+
61
+ import time
62
+ from dataclasses import dataclass, field
63
+ from pathlib import Path
64
+ from typing import ClassVar, Literal
65
+
66
+ from syncade.process import (
67
+ SubprocessError,
68
+ SubprocessNotFoundError,
69
+ SubprocessTimeoutError,
70
+ run_subprocess,
71
+ )
72
+ from syncade.test_runner_classify import (
73
+ _extract_missing_binary,
74
+ )
75
+ from syncade.worktree_env import worktree_scoped_env
76
+
77
+ TestOutcome = Literal["passed", "failed", "subprocess_error"]
78
+ """The three terminal states of a test run.
79
+
80
+ - ``"passed"`` — the test command exited 0. Tests ran and reported
81
+ no failures (by the command's own definition of "no failures";
82
+ the operator's free-form command decides what counts).
83
+ - ``"failed"`` — the test command exited non-zero (> 0). Tests ran
84
+ and reported failures; the project's own test runner is the
85
+ authority. The orchestrator folds this into the mechanical
86
+ verdict as a blocker → exit 30.
87
+ - ``"subprocess_error"`` — the subprocess itself failed: binary
88
+ missing, timeout, OS-level launch failure. Distinct from
89
+ ``"failed"`` because the operator's fix path differs (fix the
90
+ environment vs. fix the code).
91
+ """
92
+
93
+ CheckSeverity = Literal["blocking", "advisory"]
94
+
95
+
96
+ @dataclass(frozen=True) # SLOTS_OK: result shape is persisted and kept stable.
97
+ class TestRunResult:
98
+ __test__: ClassVar[bool] = False
99
+
100
+ """Outcome of one test command execution.
101
+
102
+ Mirrors the :class:`~syncade.synthesizer.SynthesizerResult` shape
103
+ so persistence can treat reviewer outcomes, synthesizer outcomes,
104
+ and test outcomes with the same vocabulary.
105
+
106
+ Attributes:
107
+ exit_code: The test command's exit code. ``0`` means tests
108
+ passed; ``> 0`` means tests failed. Sentinel ``-1`` when
109
+ the subprocess was killed (timeout) or otherwise didn't
110
+ produce a real exit code (binary missing, OS launch
111
+ failure).
112
+ outcome: One of ``"passed"`` | ``"failed"`` |
113
+ ``"subprocess_error"``. ``"passed"`` iff
114
+ ``exit_code == 0``. ``"failed"`` iff ``exit_code > 0``
115
+ (tests ran and reported failures). ``"subprocess_error"``
116
+ iff the subprocess itself failed (binary not found,
117
+ timeout, launch error) — ``exit_code`` is ``-1`` in that
118
+ case.
119
+ duration_seconds: Wall-clock duration, measured via
120
+ :func:`time.monotonic`.
121
+ stdout: Captured stdout text (whatever made it before kill on
122
+ timeout — see :func:`syncade.process.run_subprocess`'s
123
+ partial-output preservation).
124
+ stderr: Captured stderr text.
125
+ error: The exception that fired on ``subprocess_error``.
126
+ ``None`` on ``passed`` / ``failed``. Preserved so
127
+ persistence can name the exception class in
128
+ ``manifest.json``'s ``error_type`` field — same pattern
129
+ as :class:`~syncade.synthesizer.SynthesizerResult`'s
130
+ ``error`` attribute.
131
+ command: The operator-configured ``test_command`` string,
132
+ preserved verbatim so persistence can echo it back in
133
+ manifest.json and summary.md without re-fetching from
134
+ config. Empty string is allowed only on construction
135
+ from synthetic test data; production code passes the
136
+ real config value.
137
+ name: Optional mechanical-check name. ``None`` for the single
138
+ test leg; populated for ``[[checks]]`` so the check
139
+ artifacts keep their existing names.
140
+ severity: Optional mechanical-check severity. ``None`` for
141
+ the single test leg; ``"blocking"`` or ``"advisory"`` for
142
+ ``[[checks]]``.
143
+ """
144
+
145
+ exit_code: int
146
+ outcome: TestOutcome
147
+ duration_seconds: float
148
+ stdout: str
149
+ stderr: str
150
+ error: Exception | None = None
151
+ command: str = field(default="")
152
+ name: str | None = None
153
+ severity: CheckSeverity | None = None
154
+
155
+ def __post_init__(self) -> None:
156
+ """Enforce outcome ↔ exit_code consistency before the result
157
+ reaches persistence.
158
+
159
+ The contract is intentionally tight — a downstream consumer
160
+ (manifest renderer, summary.md renderer, ``_compute_exit_code``)
161
+ keys on ``outcome`` and never needs to second-guess whether
162
+ ``exit_code == 0`` and ``outcome == "failed"`` could co-occur.
163
+
164
+ Runtime validation rules:
165
+ - Reject unknown ``outcome`` values up-front (the Literal
166
+ type hint is advisory; runtime construction with
167
+ ``outcome=None`` or ``outcome="bogus"`` would silently
168
+ slip past all the rules below since none of them match).
169
+ - ``subprocess_error`` requires BOTH non-None error AND
170
+ ``exit_code == -1`` (the sentinel). Without this, a
171
+ caller could construct
172
+ ``TestRunResult(outcome="subprocess_error", exit_code=42,
173
+ error=Exception())`` and persistence would echo 42 into
174
+ manifest's ``exit_code`` field while the orchestrator's
175
+ mechanical verdict treats the result as a subprocess
176
+ error — two consumers disagreeing on what the run
177
+ actually did.
178
+ """
179
+ # Reject unknown outcomes BEFORE the per-outcome rules
180
+ # below — otherwise an out-of-vocabulary outcome would
181
+ # silently slip past every rule (none of them would match).
182
+ if self.outcome not in ("passed", "failed", "subprocess_error"):
183
+ raise ValueError( # GENERIC_ERR_OK: dataclass invariant preserves existing API.
184
+ f"TestRunResult: outcome={self.outcome!r} is not one of "
185
+ f"the three valid TestOutcome values "
186
+ f"('passed', 'failed', 'subprocess_error')"
187
+ )
188
+ if self.outcome == "passed" and self.exit_code != 0:
189
+ raise ValueError( # GENERIC_ERR_OK: dataclass invariant preserves existing API.
190
+ f"TestRunResult: outcome='passed' requires exit_code=0, "
191
+ f"got exit_code={self.exit_code}"
192
+ )
193
+ if self.outcome == "failed" and self.exit_code <= 0:
194
+ raise ValueError( # GENERIC_ERR_OK: dataclass invariant preserves existing API.
195
+ f"TestRunResult: outcome='failed' requires exit_code > 0, "
196
+ f"got exit_code={self.exit_code}"
197
+ )
198
+ if self.outcome == "subprocess_error":
199
+ if self.error is None:
200
+ raise ValueError( # GENERIC_ERR_OK: dataclass invariant preserves existing API.
201
+ "TestRunResult: outcome='subprocess_error' requires non-None error"
202
+ )
203
+ if self.exit_code != -1:
204
+ raise ValueError( # GENERIC_ERR_OK: dataclass invariant preserves existing API.
205
+ f"TestRunResult: outcome='subprocess_error' requires "
206
+ f"exit_code=-1 (the no-real-exit-code sentinel); got "
207
+ f"exit_code={self.exit_code}"
208
+ )
209
+ if self.outcome != "subprocess_error" and self.error is not None:
210
+ raise ValueError( # GENERIC_ERR_OK: dataclass invariant preserves existing API.
211
+ f"TestRunResult: error must be None when outcome != "
212
+ f"'subprocess_error' (outcome={self.outcome!r}, "
213
+ f"error={self.error!r})"
214
+ )
215
+ if (self.name is None) != (self.severity is None):
216
+ raise ValueError( # GENERIC_ERR_OK: dataclass invariant preserves existing API.
217
+ "TestRunResult: check metadata requires both name and severity"
218
+ )
219
+ if self.severity is not None and self.severity not in ("blocking", "advisory"):
220
+ raise ValueError( # GENERIC_ERR_OK: dataclass invariant preserves existing API.
221
+ "TestRunResult: severity must be 'blocking' or 'advisory'"
222
+ )
223
+
224
+
225
+ def is_blocking_check_subprocess_error(severity: str, outcome: str) -> bool:
226
+ return severity == "blocking" and outcome == "subprocess_error"
227
+
228
+
229
+ def run_tests(
230
+ *,
231
+ worktree_path: Path,
232
+ test_command: str,
233
+ timeout_seconds: float,
234
+ name: str | None = None,
235
+ severity: CheckSeverity | None = None,
236
+ ) -> TestRunResult:
237
+ """Execute ``test_command`` in ``worktree_path``, capture
238
+ stdout/stderr/rc, return a typed :class:`TestRunResult`.
239
+
240
+ The command is run via ``bash -o pipefail -c <test_command>`` so operator
241
+ pipelines fail when any component fails. The same SIGKILL-on-timeout
242
+ discipline as reviewer subprocesses applies — inherited from
243
+ :func:`syncade.process.run_subprocess`, which kills the whole process group
244
+ on timeout so descendants don't orphan.
245
+
246
+ **Note on shell=True security:** the command comes from the
247
+ operator's own ``.syncade/config.toml``, which is in their repo
248
+ and controlled by them. We are not interpolating untrusted input
249
+ into the shell string. The threat model is "operator
250
+ misconfigures their own test_command and shoots their own foot"
251
+ — not "attacker injects via test_command." Same threat model as
252
+ a Makefile or ``package.json`` ``"scripts"`` entry.
253
+
254
+ Args:
255
+ worktree_path: Path of the fresh worktree to run the test
256
+ command in. The orchestrator provisions this via
257
+ :class:`~syncade.worktree.WorktreeManager` with the
258
+ same ``strip_files`` treatment as reviewer worktrees,
259
+ so what runs sees what an external CI would see.
260
+ test_command: The operator-configured ``[loop] test_command``
261
+ string. Passed verbatim to ``bash -o pipefail -c``. Must
262
+ be a non-empty string; whitespace-only is rejected at
263
+ config-load via
264
+ :meth:`syncade.config.LoopConfig._test_command_not_whitespace`,
265
+ so callers can assume it's substantive.
266
+ timeout_seconds: Per-test-run wall-clock cap. The
267
+ orchestrator resolves this as
268
+ ``config.loop.test_timeout_seconds or
269
+ config.loop.timeout_seconds`` and passes the resolved
270
+ value here.
271
+ name: Optional mechanical-check name. The ordinary test leg
272
+ leaves this unset.
273
+ severity: Optional mechanical-check severity. Must be paired
274
+ with ``name``.
275
+
276
+ Returns:
277
+ :class:`TestRunResult` regardless of outcome. The caller
278
+ (orchestrator) inspects ``outcome`` to fold the result into
279
+ the mechanical verdict:
280
+
281
+ - ``"passed"`` → no contribution to exit code; still 0 if
282
+ synth was also clean.
283
+ - ``"failed"`` → exit 30 (treated as a blocker; the
284
+ orchestrator OR's this with
285
+ :func:`syncade.synthesis.has_active_blocker`).
286
+ - ``"subprocess_error"`` → exit 40 (same bucket as a
287
+ reviewer or synthesizer subprocess failure).
288
+ """
289
+ argv = ["bash", "-o", "pipefail", "-c", test_command]
290
+ run_start = time.monotonic()
291
+
292
+ try:
293
+ result = run_subprocess(
294
+ argv,
295
+ cwd=worktree_path,
296
+ # scope the child's Python env to the worktree so the
297
+ # authoritative test leg imports the worktree's syncade / runs the
298
+ # snapshot's pytest, NOT MAIN's via the editable-install .pth.
299
+ # The mechanical-check leg uses this same path with check metadata.
300
+ env=worktree_scoped_env(worktree_path),
301
+ timeout=timeout_seconds,
302
+ )
303
+ except SubprocessTimeoutError as exc:
304
+ # Same partial-output preservation pattern as the reviewer
305
+ # dispatcher and the synthesizer module. The partial
306
+ # stdout/stderr captured by run_subprocess before the
307
+ # SIGKILL is on the exception; surface it on the
308
+ # TestRunResult so persistence still writes the
309
+ # test-run.stdout / test-run.stderr files. Sentinel
310
+ # exit_code=-1 marks "killed before exit".
311
+ return TestRunResult(
312
+ exit_code=-1,
313
+ outcome="subprocess_error",
314
+ duration_seconds=time.monotonic() - run_start,
315
+ stdout=exc.stdout,
316
+ stderr=exc.stderr,
317
+ error=exc,
318
+ command=test_command,
319
+ name=name,
320
+ severity=severity,
321
+ )
322
+ except SubprocessNotFoundError as exc:
323
+ # ``bash`` itself missing — uncommon on supported operator systems
324
+ # but the orchestrator's error surface still needs a defined
325
+ # mapping. No partial output to preserve; the subprocess
326
+ # never started.
327
+ return TestRunResult(
328
+ exit_code=-1,
329
+ outcome="subprocess_error",
330
+ duration_seconds=time.monotonic() - run_start,
331
+ stdout="",
332
+ stderr="",
333
+ error=exc,
334
+ command=test_command,
335
+ name=name,
336
+ severity=severity,
337
+ )
338
+ except SubprocessError as exc:
339
+ # Other launch failures (bad cwd, OS error). Same shape as
340
+ # SubprocessNotFoundError — no subprocess output to
341
+ # preserve.
342
+ return TestRunResult(
343
+ exit_code=-1,
344
+ outcome="subprocess_error",
345
+ duration_seconds=time.monotonic() - run_start,
346
+ stdout="",
347
+ stderr="",
348
+ error=exc,
349
+ command=test_command,
350
+ name=name,
351
+ severity=severity,
352
+ )
353
+
354
+ # The subprocess ran to completion. Classify by exit code.
355
+ #
356
+ # The verdict architecture rests on exit codes carrying
357
+ # unambiguous operator intent:
358
+ # - Exit 30 (``outcome="failed"``) means "your code has a real
359
+ # defect, go fix the code." The multi-round loop will re-invoke the
360
+ # producer to fix code on exit 30.
361
+ # - Exit 40 (``outcome="subprocess_error"``) means "the harness
362
+ # couldn't run, go fix your environment." The loop will not continue
363
+ # on this — the operator's PATH / config / binary install is
364
+ # the fix.
365
+ #
366
+ # Three specific return-code patterns get reclassified from
367
+ # "non-zero = failed" to "subprocess_error" because they
368
+ # unambiguously signal harness/environment problems rather
369
+ # than test failures:
370
+ #
371
+ # - **127** (POSIX "command not found"): bash pipefail execution
372
+ # returned this because the operator's command references a binary
373
+ # that isn't on PATH. This is a subprocess_error, and the classifier
374
+ # parses the actual missing binary out of stderr for the error message.
375
+ # - **126** (POSIX "command found but not executable"): the
376
+ # shell located the binary but the OS refused to exec it
377
+ # (permission denied, wrong arch, broken shebang). Same
378
+ # class of environmental problem as 127.
379
+ # - **Negative return code** (signal-killed): the subprocess
380
+ # was terminated by a signal before it could exit normally.
381
+ # This isn't the test reporting failure — the test process
382
+ # never got to decide. Examples: ``kill -TERM $$`` from
383
+ # inside the script, OOM-killer, parent-shell death.
384
+ #
385
+ # All three preserve stdout/stderr so the operator inspecting
386
+ # test-run.stderr sees what the shell / OS said about the
387
+ # failure. exit_code on the TestRunResult is -1 (the
388
+ # subprocess_error sentinel) — the real return code lives in
389
+ # the synthesized exception's message and in the captured
390
+ # streams.
391
+ #
392
+ # Risk of false positive: a hypothetical test framework that
393
+ # exits 126/127 on genuine test failure would be
394
+ # reclassified. No major framework does (pytest 0-5, npm 1,
395
+ # jest 1). Documented escape hatch: wrap the command to
396
+ # remap, e.g. ``bash -o pipefail -c 'runner; rc=$?; case $rc in 126|127)
397
+ # exit 1;; *) exit $rc;; esac'``.
398
+ rc = result.returncode
399
+ if rc == 127:
400
+ # Prefer to name the actual missing binary (extracted from stderr) over
401
+ # the full command string. Fallback to the full command if parse fails.
402
+ missing = _extract_missing_binary(result.stderr) or test_command
403
+ synthesized_error: Exception = SubprocessNotFoundError(missing)
404
+ return TestRunResult(
405
+ exit_code=-1,
406
+ outcome="subprocess_error",
407
+ duration_seconds=result.duration_seconds,
408
+ stdout=result.stdout,
409
+ stderr=result.stderr,
410
+ error=synthesized_error,
411
+ command=test_command,
412
+ name=name,
413
+ severity=severity,
414
+ )
415
+ if rc == 126:
416
+ synthesized_error = SubprocessError(
417
+ f"test_command found but not executable (bash/shell exit 126): "
418
+ f"{test_command!r}. The OS refused to exec the named "
419
+ f"binary — common causes: permission denied (chmod +x), "
420
+ f"wrong architecture, or a broken shebang. Captured "
421
+ f"shell stderr: {result.stderr.strip()[:200]!r}"
422
+ )
423
+ return TestRunResult(
424
+ exit_code=-1,
425
+ outcome="subprocess_error",
426
+ duration_seconds=result.duration_seconds,
427
+ stdout=result.stdout,
428
+ stderr=result.stderr,
429
+ error=synthesized_error,
430
+ command=test_command,
431
+ name=name,
432
+ severity=severity,
433
+ )
434
+ if rc < 0:
435
+ # subprocess.Popen returns negative rc when the child was
436
+ # terminated by a signal. ``-N`` means signal N (so -15 =
437
+ # SIGTERM, -9 = SIGKILL — though SIGKILL from the orch's
438
+ # own timeout takes the SubprocessTimeoutError path
439
+ # above, not this one). Signal-killed isn't a test
440
+ # verdict; it's the test process never getting to
441
+ # exit.
442
+ synthesized_error = SubprocessError(
443
+ f"test_command terminated by signal {-rc} before it "
444
+ f"could exit normally (raw subprocess returncode={rc}). "
445
+ f"This is a harness/environment problem, not a test "
446
+ f"verdict. Captured shell stderr: "
447
+ f"{result.stderr.strip()[:200]!r}"
448
+ )
449
+ return TestRunResult(
450
+ exit_code=-1,
451
+ outcome="subprocess_error",
452
+ duration_seconds=result.duration_seconds,
453
+ stdout=result.stdout,
454
+ stderr=result.stderr,
455
+ error=synthesized_error,
456
+ command=test_command,
457
+ name=name,
458
+ severity=severity,
459
+ )
460
+
461
+ if rc == 0:
462
+ outcome: TestOutcome = "passed"
463
+ else:
464
+ outcome = "failed"
465
+
466
+ return TestRunResult(
467
+ exit_code=rc,
468
+ outcome=outcome,
469
+ duration_seconds=result.duration_seconds,
470
+ stdout=result.stdout,
471
+ stderr=result.stderr,
472
+ error=None,
473
+ command=test_command,
474
+ name=name,
475
+ severity=severity,
476
+ )