opencode-swarm 7.136.5 → 7.137.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.opencode/skills/swarm-pr-review/SKILL.md +35 -3
- package/.opencode/skills/swarm-pr-review/references/prompt-templates.md +7 -3
- package/dist/agents/agent-output-schema.d.ts +2 -2
- package/dist/background/candidate-contract.d.ts +14 -0
- package/dist/cli/{config-doctor-n3hatm9d.js → config-doctor-b9r9rt2c.js} +2 -2
- package/dist/cli/{core-jpjk2qvt.js → core-6w5sd5bc.js} +1 -1
- package/dist/cli/{curation-policy-c4g08bg1.js → curation-policy-ftx2emfw.js} +7 -6
- package/dist/cli/{curator-jc5cc980.js → curator-1sxxtbq7.js} +25 -25
- package/dist/cli/{curator-llm-factory-atfnhbn6.js → curator-llm-factory-yhqbkc5w.js} +25 -25
- package/dist/cli/{evidence-summary-service-nse06dqr.js → evidence-summary-service-85gw9p2m.js} +7 -7
- package/dist/cli/{gate-evidence-kdygjr56.js → gate-evidence-s2kytsrj.js} +4 -4
- package/dist/cli/{guardrail-explain-v6f8efve.js → guardrail-explain-qh1y8jsz.js} +26 -26
- package/dist/cli/{guardrail-log-0mrq14g9.js → guardrail-log-84zffdv3.js} +3 -3
- package/dist/cli/{hive-promoter-wyejhz8b.js → hive-promoter-hpe29xrz.js} +25 -25
- package/dist/cli/{index-eeb35dnx.js → index-1qn2gchc.js} +3 -2
- package/dist/cli/{index-45y0y3xh.js → index-1zrb3vt2.js} +7 -7
- package/dist/cli/{index-d2sf9an1.js → index-5r6kkhd7.js} +4 -4
- package/dist/cli/{index-ne28wyyc.js → index-5wjw05dy.js} +2 -2
- package/dist/cli/{index-xvcctb2t.js → index-7e60c7ay.js} +27 -27
- package/dist/cli/{index-7t3vjw5e.js → index-877ah4kx.js} +1 -1
- package/dist/cli/{index-tncy55bp.js → index-8ehyt3rx.js} +1 -1
- package/dist/cli/{index-hbzh7yyz.js → index-9vqxf9pd.js} +633 -209
- package/dist/cli/{index-6kd0ezgd.js → index-a3gpeb4f.js} +2 -2
- package/dist/cli/{index-nfm9f10v.js → index-b2s3t517.js} +5 -3
- package/dist/cli/{index-8f4346kf.js → index-bxk0a1r9.js} +2 -2
- package/dist/cli/{index-pyz1p8qv.js → index-d67810xe.js} +1 -1
- package/dist/cli/{index-akc6rp3x.js → index-d8rc8c9y.js} +2 -2
- package/dist/cli/{index-hh34tyv7.js → index-dxatvs3f.js} +1 -1
- package/dist/cli/{index-bqan4y2y.js → index-e0nvge1c.js} +2 -2
- package/dist/cli/{index-f5qs217h.js → index-em10pmph.js} +1 -1
- package/dist/cli/{index-j7ja83v5.js → index-gd47gp2q.js} +1 -1
- package/dist/cli/{index-w6j1n5az.js → index-gv00q5gc.js} +1 -1
- package/dist/cli/index-h5sbn1zr.js +1670 -0
- package/dist/cli/{index-vm05set2.js → index-k8y41t4y.js} +4 -4
- package/dist/cli/{index-tr0ctbhq.js → index-m4s91910.js} +5 -5
- package/dist/cli/{index-b3jfcptk.js → index-n3h6rmbe.js} +1 -1
- package/dist/cli/{index-xp2pfbpq.js → index-pbe51g7h.js} +1 -1
- package/dist/cli/{index-kym0cctr.js → index-qy3rgmkc.js} +1 -1
- package/dist/cli/{index-d61h3f3h.js → index-s3j9fsab.js} +2 -2
- package/dist/cli/{index-qhbr4h5n.js → index-t2mv25yf.js} +6 -6
- package/dist/cli/{index-84se71v1.js → index-t5099rx9.js} +3 -3
- package/dist/cli/{index-zzhyws9g.js → index-vft1pg34.js} +1 -1
- package/dist/cli/{index-yj61bped.js → index-zxf5w0jj.js} +3 -3
- package/dist/cli/index.js +25 -25
- package/dist/cli/{knowledge-escalator-2xe8z24p.js → knowledge-escalator-3ds615yq.js} +8 -7
- package/dist/cli/{knowledge-events-wecrshsx.js → knowledge-events-3c0g2xbm.js} +6 -5
- package/dist/cli/{knowledge-link-zr40rnwr.js → knowledge-link-t94b79qe.js} +5 -4
- package/dist/cli/{knowledge-store-9f74whxw.js → knowledge-store-pq43990j.js} +6 -5
- package/dist/cli/{knowledge-validator-4ng7x9dr.js → knowledge-validator-gepda374.js} +9 -8
- package/dist/cli/{pending-delegations-0h5b18p7.js → pending-delegations-kshh66s2.js} +3 -3
- package/dist/cli/{pr-subscriptions-jn0h047q.js → pr-subscriptions-5fpzz4f3.js} +3 -3
- package/dist/cli/{runner-deeswadt.js → runner-9rqc8y3m.js} +5 -5
- package/dist/cli/{scan-cursor-yzj9n225.js → scan-cursor-5hq37qdg.js} +7 -6
- package/dist/cli/{schema-cr1nr1w0.js → schema-mrbyt7dg.js} +1 -1
- package/dist/cli/{scope-persistence-5xc9ntdh.js → scope-persistence-sjz200dt.js} +4 -4
- package/dist/cli/{skill-generator-ybncj9et.js → skill-generator-3gt8em6v.js} +10 -9
- package/dist/cli/{telemetry-6678gya0.js → telemetry-h7h7f0ez.js} +2 -1
- package/dist/cli/{worktree-collision-ownership-wt7cc850.js → worktree-collision-ownership-xzwab4xt.js} +3 -3
- package/dist/config/constants.d.ts +1 -1
- package/dist/config/evidence-schema.d.ts +103 -103
- package/dist/config/plan-schema.d.ts +10 -10
- package/dist/config/schema.d.ts +8 -8
- package/dist/consensus/contracts.d.ts +3 -3
- package/dist/evaluation/index.d.ts +1 -0
- package/dist/evaluation/model-dispatcher.d.ts +19 -1
- package/dist/evaluation/pr-review-recovery.d.ts +41 -0
- package/dist/evaluation/public-api.d.ts +2 -0
- package/dist/evaluation/runner.d.ts +9 -0
- package/dist/hooks/hive-promoter.d.ts +1 -1
- package/dist/hooks/pr-workflow-gate.d.ts +69 -0
- package/dist/index.d.ts +2 -0
- package/dist/index.js +225 -225
- package/dist/observability/catalog.d.ts +84 -0
- package/dist/observability/envelope.d.ts +401 -0
- package/dist/observability/ids.d.ts +97 -0
- package/dist/observability/index.d.ts +58 -0
- package/dist/observability/legacy.d.ts +68 -0
- package/dist/observability/observe.d.ts +112 -0
- package/dist/observability/otel-mapping.d.ts +48 -0
- package/dist/observability/relationships.d.ts +44 -0
- package/dist/observability/sampling.d.ts +77 -0
- package/dist/summaries/schema.d.ts +2 -2
- package/dist/telemetry.d.ts +4 -1
- package/dist/tools/dispatch-lanes.d.ts +1 -0
- package/evaluation-fixtures/pr-review-recovery/baseline/SKILL.md +1823 -0
- package/evaluation-fixtures/pr-review-recovery/baseline-manifest.json +1 -0
- package/evaluation-fixtures/pr-review-recovery/environment/score-recovery.cjs +135 -0
- package/evaluation-fixtures/pr-review-recovery/instruction.md +40 -0
- package/evaluation-fixtures/pr-review-recovery/scorer-fixtures/blind-full-wave-retry/model-output.json +1 -0
- package/evaluation-fixtures/pr-review-recovery/scorer-fixtures/contradictory/model-output.json +1 -0
- package/evaluation-fixtures/pr-review-recovery/scorer-fixtures/copied-contract/model-output.json +1 -0
- package/evaluation-fixtures/pr-review-recovery/scorer-fixtures/critical-without-harm/model-output.json +1 -0
- package/evaluation-fixtures/pr-review-recovery/scorer-fixtures/invented-harm/model-output.json +1 -0
- package/evaluation-fixtures/pr-review-recovery/scorer-fixtures/malformed/model-output.json +1 -0
- package/evaluation-fixtures/pr-review-recovery/scorer-fixtures/missing-reproduction/model-output.json +1 -0
- package/evaluation-fixtures/pr-review-recovery/scorer-fixtures/parser-only/model-output.json +1 -0
- package/evaluation-fixtures/pr-review-recovery/scorer-fixtures/passing/model-output.json +1 -0
- package/evaluation-fixtures/pr-review-recovery/scorer-fixtures/premature-systemic-claim/model-output.json +1 -0
- package/evaluation-fixtures/pr-review-recovery/scorer-fixtures/profile-b-fallback/model-output.json +1 -0
- package/evaluation-fixtures/pr-review-recovery/scorer-fixtures/underclaimed-harm/model-output.json +1 -0
- package/evaluation-fixtures/pr-review-recovery/scorer-fixtures/unsupported-severity/model-output.json +1 -0
- package/package.json +3 -2
- package/dist/cli/index-cze4bq1x.js +0 -358
|
@@ -0,0 +1,1823 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: swarm-pr-review
|
|
3
|
+
audience: swarm-plugin
|
|
4
|
+
description: Run a graph-guided, tool-augmented PR review using context packing, parallel exploration, mandatory repository-agnostic risk-family coverage with dispatch scaled to diff size and risk, independent reviewer validation, critic challenge, and metrics writeback. Use for deep pull request review with low false-positive tolerance and high recall in any repository, on any agent harness (structured lane controller, native parallel subagents, or single-context sequential passes).
|
|
5
|
+
disable-model-invocation: true
|
|
6
|
+
---
|
|
7
|
+
|
|
8
|
+
# /swarm-pr-review
|
|
9
|
+
|
|
10
|
+
Run a structured, high-confidence PR review that maximizes valid findings without flooding the user with unvalidated noise.
|
|
11
|
+
|
|
12
|
+
The review ladder is:
|
|
13
|
+
|
|
14
|
+
**Scope → obligations → context pack → deterministic signals → parallel explorers → repository-agnostic risk-family coverage (dispatch scaled by depth tier) → independent reviewer validation → critic challenge → grouped synthesis → metrics / knowledge writeback.**
|
|
15
|
+
|
|
16
|
+
## Handoff To PR Feedback
|
|
17
|
+
|
|
18
|
+
Use `../swarm-pr-feedback/SKILL.md` instead of this skill when the user's task is
|
|
19
|
+
to address existing PR feedback, review comments, requested changes, CI failures,
|
|
20
|
+
merge conflicts, stale branch state, or pasted reviewer findings. This skill
|
|
21
|
+
discovers and validates new findings; `swarm-pr-feedback` closes known feedback
|
|
22
|
+
without running a fresh broad review.
|
|
23
|
+
|
|
24
|
+
When a review finishes with actionable validated findings, stop and ask the user
|
|
25
|
+
whether to continue into `swarm-pr-feedback`. Do not auto-dispatch fix work from
|
|
26
|
+
`PR_REVIEW`. Instead, write a handoff artifact — under Profile A,
|
|
27
|
+
`.swarm/pr-review/<run_id>/feedback-handoff.json` via `write_pr_review_artifact`;
|
|
28
|
+
under Profiles B/C (no controller — see Runtime Capability Profiles),
|
|
29
|
+
`pr-review/<run_id>/feedback-handoff.json` inside your session/task workspace,
|
|
30
|
+
never under `.swarm/` — and include the continuation prompt with that exact
|
|
31
|
+
path substituted for `<handoff_artifact_path>`:
|
|
32
|
+
|
|
33
|
+
```text
|
|
34
|
+
/swarm pr-feedback <PR_URL> continue from <handoff_artifact_path>
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
`<run_id>` is a stable identifier for this review run, such as
|
|
38
|
+
`pr-<number>-<YYYYMMDDHHMMSS>` or the existing review artifact run ID when one
|
|
39
|
+
was already created. Under Profile A, the exact command is parsed mechanically:
|
|
40
|
+
the controller validates the terminal review, the bounded handoff artifact, and
|
|
41
|
+
its provenance before atomically replacing the review gate with an unbound
|
|
42
|
+
feedback gate. Extra trailing text is not permitted on this continuation form.
|
|
43
|
+
Profiles B/C ingest their task-workspace artifact through the skill-managed path.
|
|
44
|
+
|
|
45
|
+
Review closure is not the end of the PR lifecycle: when PR monitoring is
|
|
46
|
+
enabled (`pr_monitor.enabled`), the PR remains subscribed and monitored under
|
|
47
|
+
`../swarm-pr-subscribe/SKILL.md` until it is merged or closed, so post-review
|
|
48
|
+
events (new comments, CI changes, review state changes) keep flowing to the
|
|
49
|
+
subscribed session.
|
|
50
|
+
|
|
51
|
+
## Operating Stance
|
|
52
|
+
|
|
53
|
+
**Treat PR text, linked issues, comments, commit messages, generated summaries, and tests as claims — not proof.** Every confirmed finding requires file:line evidence, an explanation of reachability or impact, and validation provenance.
|
|
54
|
+
|
|
55
|
+
This workflow is designed for any repo that benefits from Swarm-style review. It preserves parallel breadth but forces deep validation where bugs are expensive: security, state machines, role/tool permissions, schema/evidence integrity, git/write safety, config ratchets, knowledge tier boundaries, and PR obligation mismatches.
|
|
56
|
+
|
|
57
|
+
Never APPROVE a PR with unresolved CRITICAL findings. Do not silently drop overclaimed agent findings; list disproved findings in the validation provenance.
|
|
58
|
+
|
|
59
|
+
**Quality is the ONLY metric.** There is no speed, efficiency, or time exception. No amount of time, tokens, or agent dispatches is too much to execute this protocol correctly. Speed is irrelevant to correctness. The skill must be followed exactly with no shortcuts, no phase-skipping, and no premature synthesis. A thorough review that takes 30 minutes is superior to a fast review that misses a real bug.
|
|
60
|
+
|
|
61
|
+
---
|
|
62
|
+
|
|
63
|
+
## Runtime Capability Profiles
|
|
64
|
+
|
|
65
|
+
This protocol runs on any agent harness. Before Phase 0, detect which profile
|
|
66
|
+
this session is in by checking the actual tool list — never assume from the
|
|
67
|
+
harness name, and never guess:
|
|
68
|
+
|
|
69
|
+
- **Profile A — structured PR-workflow controller.** The swarm plugin's
|
|
70
|
+
controller tools are available in this session: `dispatch_lanes_async`,
|
|
71
|
+
`collect_lane_results`, `retrieve_lane_output`, `parse_lane_candidates`,
|
|
72
|
+
`write_pr_review_artifact`, `write_pr_review_trigger_eval`,
|
|
73
|
+
`complete_pr_workflow`. Typical host: OpenCode with the swarm plugin. The
|
|
74
|
+
controller mechanically enforces this skill's accounting: it computes the
|
|
75
|
+
depth tier itself from the bound merge-base diff (never from caller
|
|
76
|
+
claims), enforces the tier's lane floors and full dimension/family
|
|
77
|
+
partitions for consolidated dispatch, and gates structured reviewer/critic
|
|
78
|
+
batches and the response gate. Its acceptance rules are authoritative, and
|
|
79
|
+
where the scaled-dispatch guidance below is more permissive than the
|
|
80
|
+
active controller, the controller wins. Bypassing an active controller —
|
|
81
|
+
blocking `dispatch_lanes`, direct Task/agent dispatch, prose verdicts — is
|
|
82
|
+
BLOCKED.
|
|
83
|
+
- **Profile B — native parallel subagents, no controller.** The controller
|
|
84
|
+
tools are absent, but the harness can spawn independent fresh-context
|
|
85
|
+
subagents (for example Claude Code's `Agent`/`Task` tool, or the native
|
|
86
|
+
subagent mechanisms in Codex and ZCode). Run the same phases, role
|
|
87
|
+
boundaries, row contracts, and join barriers; you are the accounting layer
|
|
88
|
+
the controller would otherwise be: bind the exact `pr_head_sha` in every
|
|
89
|
+
lane prompt, record per-lane provenance (lane id, head SHA) on every ledger
|
|
90
|
+
row, settle every lane before the next phase begins, and persist ledgers to
|
|
91
|
+
files in your harness's session/task workspace. Never write runtime
|
|
92
|
+
artifacts under `.swarm/` — that directory belongs to the plugin controller.
|
|
93
|
+
- **Profile C — single context, no subagents.** The harness cannot spawn
|
|
94
|
+
independent subagents in-session. Execute the same phases as strictly
|
|
95
|
+
separated sequential passes — candidate generation, then reviewer
|
|
96
|
+
validation, then critic challenge — re-deriving rather than restating
|
|
97
|
+
earlier reasoning in each pass, with the same ledger rows and per-family
|
|
98
|
+
attestations. Disclose in the validation provenance that reviewer/critic
|
|
99
|
+
independence was procedural (separate passes in one context), not
|
|
100
|
+
contextual.
|
|
101
|
+
|
|
102
|
+
| Harness (typical) | Profile | Lane dispatch | Ledger persistence | Completion gate |
|
|
103
|
+
|---|---|---|---|---|
|
|
104
|
+
| OpenCode + swarm plugin | A | `dispatch_lanes_async` / `collect_lane_results` | `write_pr_review_artifact`, `write_pr_review_trigger_eval` | `complete_pr_workflow` |
|
|
105
|
+
| Claude Code | B | parallel `Agent`/`Task` subagents | ledger files in the session task workspace | Pre-Synthesis Gate checklist |
|
|
106
|
+
| OpenAI Codex | B | parallel subagents (fresh context) | ledger files in working notes | Pre-Synthesis Gate checklist |
|
|
107
|
+
| ZCode | B | parallel subagents (fresh context) | ledger files in working notes | Pre-Synthesis Gate checklist |
|
|
108
|
+
|
|
109
|
+
Verify each row against your own current tool list before relying on it; a
|
|
110
|
+
harness may gain or lose capabilities between versions. OpenCode, Claude Code,
|
|
111
|
+
Codex, and ZCode can all spawn fresh-context subagents in current versions —
|
|
112
|
+
run Profile B wherever the session actually exposes that capability, and
|
|
113
|
+
reserve Profile C for sessions that genuinely lack a subagent mechanism; never
|
|
114
|
+
assign a harness to Profile C by name alone. The absence of the controller is
|
|
115
|
+
NOT a BLOCKED condition — Profiles B and C are first-class execution paths, not degraded fallbacks.
|
|
116
|
+
BLOCKED is reserved for bypassing an active controller and for coverage gaps
|
|
117
|
+
that remain unclosable after bounded retries on any profile.
|
|
118
|
+
|
|
119
|
+
---
|
|
120
|
+
|
|
121
|
+
## Review Modes
|
|
122
|
+
|
|
123
|
+
### Default layered workflow
|
|
124
|
+
|
|
125
|
+
Always run the default layered workflow (mechanically enforced under Profile A). Explorers produce only candidates. The orchestrator does not confirm or disprove candidates.
|
|
126
|
+
|
|
127
|
+
### Council mode — opt in only
|
|
128
|
+
|
|
129
|
+
Council mode applies only when the user explicitly says one of:
|
|
130
|
+
|
|
131
|
+
- `council`
|
|
132
|
+
- `independent review`
|
|
133
|
+
- `N-agent review`
|
|
134
|
+
- `/council`
|
|
135
|
+
- `[COUNCIL MODE]`
|
|
136
|
+
- `[MODE: PR_REVIEW … council=true]`
|
|
137
|
+
- `assume all work is wrong`
|
|
138
|
+
|
|
139
|
+
Council mode supplements the default mechanical workflow; it never replaces or weakens it. Even when council mode is triggered, first complete the base-dimension coverage (the tier-floored base dispatch under Profile A — the exact-six wave at depth tier L), micro-lane ledger persistence, and every repository-agnostic risk-family evaluation at the same exact `pr_head_sha`. Route supplementary council output into the candidate ledger before independent reviewer classification. If the council request arrives after classification has begun, run the council as an additional candidate pass and dispatch a new structured reviewer batch for those candidates before synthesis.
|
|
140
|
+
|
|
141
|
+
---
|
|
142
|
+
|
|
143
|
+
## Anti-Self-Review Rule
|
|
144
|
+
|
|
145
|
+
The main thread / orchestrator MUST NOT classify, confirm, disprove, or judge explorer candidates in the default workflow.
|
|
146
|
+
|
|
147
|
+
The orchestrator may:
|
|
148
|
+
|
|
149
|
+
- determine scope,
|
|
150
|
+
- build or request the context pack,
|
|
151
|
+
- launch explorers and the full risk-family micro coverage (every family evaluated; lane count per depth tier and profile),
|
|
152
|
+
- extract candidates from lane artifacts via `parse_lane_candidates` (Profile A) or by collecting the structured `[CANDIDATE]` rows from lane reports (Profiles B/C),
|
|
153
|
+
- filter, group, and chunk candidates for reviewer dispatch,
|
|
154
|
+
- route candidates to reviewers,
|
|
155
|
+
- route reviewer-confirmed findings to critics,
|
|
156
|
+
- group validated findings,
|
|
157
|
+
- prepare the final report.
|
|
158
|
+
|
|
159
|
+
The orchestrator MUST NOT:
|
|
160
|
+
|
|
161
|
+
- re-read a candidate's target code to decide if it is valid,
|
|
162
|
+
- silently downgrade or discard an explorer candidate,
|
|
163
|
+
- treat tool output as a confirmed finding,
|
|
164
|
+
- report a finding that no reviewer validated,
|
|
165
|
+
- classify or judge candidates based on preview text alone — always use the structured parser output (Profile A) or the verbatim-collected `[CANDIDATE]` rows (Profiles B/C).
|
|
166
|
+
|
|
167
|
+
If the orchestrator catches itself validating code, it must stop and delegate validation to a reviewer subagent.
|
|
168
|
+
|
|
169
|
+
Exception: in explicit Council mode only, the main thread may act as the independent reviewer as described in the Council Mode section. Prefer a reviewer subagent when available.
|
|
170
|
+
|
|
171
|
+
---
|
|
172
|
+
|
|
173
|
+
## Scope Detection
|
|
174
|
+
|
|
175
|
+
Determine review scope using this priority:
|
|
176
|
+
|
|
177
|
+
1. explicit user-provided PR URL, PR number, commit, branch, or file scope,
|
|
178
|
+
2. current feature branch diff vs the remote-tracking base ref (`origin/main`,
|
|
179
|
+
`origin/master`; a local `main`/`master` only as a last resort — it is only as
|
|
180
|
+
fresh as the last fetch and yields a different merge base),
|
|
181
|
+
3. staged changes,
|
|
182
|
+
4. latest commit,
|
|
183
|
+
5. user-specified files or directories.
|
|
184
|
+
|
|
185
|
+
Record:
|
|
186
|
+
|
|
187
|
+
- base ref,
|
|
188
|
+
- head ref,
|
|
189
|
+
- commit range,
|
|
190
|
+
- changed files,
|
|
191
|
+
- deleted files,
|
|
192
|
+
- generated files,
|
|
193
|
+
- lockfiles,
|
|
194
|
+
- test files,
|
|
195
|
+
- docs/config/schema files,
|
|
196
|
+
- whether the working tree is dirty.
|
|
197
|
+
|
|
198
|
+
If scope cannot be determined, review the narrowest safe scope available and state the limitation.
|
|
199
|
+
|
|
200
|
+
### Pre-flight git ref availability
|
|
201
|
+
|
|
202
|
+
Before launching explorers (Phase 3), perform this exact standalone sequence:
|
|
203
|
+
|
|
204
|
+
1. Resolve and retain the authoritative full `pr_head_sha` from PR metadata.
|
|
205
|
+
2. Verify the working tree is clean with `git status --porcelain`. If it is
|
|
206
|
+
dirty at all — tracked changes, untracked files, or both — call
|
|
207
|
+
`prepare_pr_workflow_checkout` (Profile A). The tool supports
|
|
208
|
+
self-discovery: call it with no `paths` argument to auto-discover and
|
|
209
|
+
preserve every dirty path in one step, including untracked files; an
|
|
210
|
+
already-clean tree returns a no-op (nothing is stashed, no receipt is
|
|
211
|
+
written). Pass explicit `paths` only when you already have the exact,
|
|
212
|
+
bounded list of dirty tracked files and want the older exact-match
|
|
213
|
+
contract. Without the controller (Profiles B/C), do not blind-stash over
|
|
214
|
+
dirty state: surface tracked changes to the user, or abort. Do not issue
|
|
215
|
+
`git stash` through shell.
|
|
216
|
+
Treat the controller's Git-state result as final for this attempt: `clean`
|
|
217
|
+
proceeds, `stashable` permits exactly one checkout-preparation call, and
|
|
218
|
+
`recovery-required` or `indeterminate` means report the typed
|
|
219
|
+
`required_action`, abort/clear any already-active gate, and stop. Retry only
|
|
220
|
+
when the controller explicitly returns `retryable: true`; never fight an
|
|
221
|
+
unmerged index or in-progress Git operation with repeated stash attempts.
|
|
222
|
+
3. Fetch the PR head as one standalone command, for example
|
|
223
|
+
`git fetch origin refs/pull/<N>/head`. Do not compose fetch and checkout.
|
|
224
|
+
3a. Fetch the base branch as its own standalone command, for example
|
|
225
|
+
`git fetch origin main`. Skipping this is the single most common cause of a
|
|
226
|
+
rejected dispatch: the merge base is recomputed against `base_ref`, and a
|
|
227
|
+
local `main`/`refs/heads/main` that was never refreshed resolves to a
|
|
228
|
+
different commit than `origin/main` for the same `base_sha`.
|
|
229
|
+
4. Prove the full commit exists locally with
|
|
230
|
+
`git cat-file -e <full_pr_head_sha>^{commit}`.
|
|
231
|
+
5. Check out the exact PR filesystem with
|
|
232
|
+
`git switch --detach <full_pr_head_sha>`. Do not use `--track FETCH_HEAD`:
|
|
233
|
+
`FETCH_HEAD` is not a remote-tracking branch.
|
|
234
|
+
6. Confirm `git rev-parse HEAD` equals the full `pr_head_sha`, bind that exact
|
|
235
|
+
head (Profile A: through the first PR-review controller call; Profiles B/C:
|
|
236
|
+
record it at the top of the findings ledger and repeat it in every lane
|
|
237
|
+
prompt), and finish this before dispatching explorer lanes.
|
|
238
|
+
|
|
239
|
+
Explorer agents read files from the working tree, not from git history. Passing
|
|
240
|
+
the commit range in a prompt cannot substitute for this checkout because
|
|
241
|
+
`Read` / `Glob` / `Grep` operate on the filesystem.
|
|
242
|
+
- Explicitly pass the verified merge-base range (`base_sha...pr_head_sha`) in every explorer delegation so explorers inspect exactly the bound PR diff. Include `base_ref` only as the live ref used to recompute `base_sha`; do not substitute a two-dot branch-tip range.
|
|
243
|
+
|
|
244
|
+
If refs cannot be fetched or checked out, state the limitation in the context pack.
|
|
245
|
+
|
|
246
|
+
### Shell rules under the PR_REVIEW gate
|
|
247
|
+
|
|
248
|
+
The gate accepts one command per tool call — never compose commands with
|
|
249
|
+
`&&`, `;`, `|`, redirects (`>`, `>>`, `<`), or `$(...)`/`` ` `` substitution.
|
|
250
|
+
A single leading `cd <dir> &&` prefix and a trailing `2>&1` suffix are
|
|
251
|
+
tolerated, but only on read-only commands. State-transition verbs — `git
|
|
252
|
+
fetch`, `git checkout`, `git switch`, `git branch`, and `gh pr checkout` —
|
|
253
|
+
must always run bare: no `cd` prefix, no `2>&1` suffix.
|
|
254
|
+
|
|
255
|
+
Allowed read-only `git` subcommands: `status`, `log`, `show`, `diff`,
|
|
256
|
+
`rev-parse`, `merge-base`, `ls-files`, `grep`, `blame`, `cat-file`,
|
|
257
|
+
`for-each-ref`, `branch --list` (listing only — mutation flags are blocked),
|
|
258
|
+
`remote -v`, and `config --get`.
|
|
259
|
+
|
|
260
|
+
Prefer tools over raw shell for state that a single read-only command cannot
|
|
261
|
+
cover cleanly:
|
|
262
|
+
|
|
263
|
+
- `pr_workflow_status` — observe local HEAD, branch, dirty-file state,
|
|
264
|
+
remotes, and gate state in one read-only call.
|
|
265
|
+
- `gh_evidence` — bounded PR/issue/run metadata without a shell round-trip.
|
|
266
|
+
- If `gh` is not installed, the web fetch tool against the equivalent
|
|
267
|
+
`api.github.com` REST URL is the degraded read-only path.
|
|
268
|
+
|
|
269
|
+
## Phase 0A: Existing PR Signal Ingestion
|
|
270
|
+
|
|
271
|
+
When reviewing a PR, ingest and triage every existing signal BEFORE starting
|
|
272
|
+
Phase 0. These are candidate generators and obligation sources, not
|
|
273
|
+
pre-confirmed findings.
|
|
274
|
+
|
|
275
|
+
### PR title and body compliance check
|
|
276
|
+
|
|
277
|
+
Before deeper analysis, discover whether the repository defines a PR
|
|
278
|
+
publication contract (for example a local `commit-pr` skill, `CONTRIBUTING`
|
|
279
|
+
guidance, a PR template, or a CI check such as `pr-standards`). If it does,
|
|
280
|
+
verify the PR against that contract and record any gap as an advisory ledger
|
|
281
|
+
item. If it does not, do not invent opencode-swarm-specific title/body
|
|
282
|
+
sections; still verify that the PR text is not misleading about what the diff
|
|
283
|
+
does or proves.
|
|
284
|
+
|
|
285
|
+
At minimum, check:
|
|
286
|
+
|
|
287
|
+
- required title/body/linked-issue structure from the discovered repository
|
|
288
|
+
contract,
|
|
289
|
+
- issue-closing, migration, release-note, invariant, or test-plan claims made
|
|
290
|
+
in the PR text,
|
|
291
|
+
- whether those claims are supported by the actual diff and the current issue
|
|
292
|
+
state.
|
|
293
|
+
|
|
294
|
+
**Issue-closing claim-integrity check:** if the PR body uses an issue-closing
|
|
295
|
+
keyword such as `Closes #<issue-number>`, verify (a) the issue is currently open
|
|
296
|
+
(`gh issue view <N> --json state` when the host is GitHub), and (b) the diff
|
|
297
|
+
addresses the issue's acceptance criteria (read the issue, map each criterion
|
|
298
|
+
to changed files/symbols, and inspect the diff for those areas). If the issue
|
|
299
|
+
is already closed by another merged PR, do NOT re-close it — the duplicate
|
|
300
|
+
closing reference is misleading. If the issue is open but the diff does not
|
|
301
|
+
address the acceptance criteria, mark the claim as `UNVERIFIED — claim
|
|
302
|
+
integrity` in the validation provenance and surface the unresolved gap to the
|
|
303
|
+
user before synthesis.
|
|
304
|
+
|
|
305
|
+
Contract non-compliance is a ledger item (advisory unless the repository
|
|
306
|
+
explicitly makes it blocking). If the PR is from an external contributor, note
|
|
307
|
+
the compliance gap for the maintainer to address before merge.
|
|
308
|
+
|
|
309
|
+
This intake includes:
|
|
310
|
+
|
|
311
|
+
- review comments, review summaries, requested changes, and bot findings,
|
|
312
|
+
- CI/check failures, annotations, and relevant logs,
|
|
313
|
+
- mergeability/conflicts, `mergeStateStatus`, and stale/base-drift state,
|
|
314
|
+
- PR body claims, linked issues, acceptance criteria, and test-plan claims,
|
|
315
|
+
- commit messages and app/bot commits on the PR branch.
|
|
316
|
+
|
|
317
|
+
When thread resolution state matters, prefer GraphQL review-thread inspection.
|
|
318
|
+
If GraphQL is unavailable, keep the signal and mark
|
|
319
|
+
`resolution_state: UNKNOWN`; do not drop it from scope.
|
|
320
|
+
|
|
321
|
+
### Step 1 — Fetch all PR feedback surfaces
|
|
322
|
+
|
|
323
|
+
The commands below are GitHub examples. On GitLab, Bitbucket, Gerrit, or
|
|
324
|
+
another code host, use the host's API/connector/CLI to enumerate the same full
|
|
325
|
+
surface, including pagination and unresolved-thread state. Host choice never
|
|
326
|
+
reduces the intake ledger.
|
|
327
|
+
|
|
328
|
+
```bash
|
|
329
|
+
# Issue comments (general PR thread)
|
|
330
|
+
gh api --paginate repos/{owner}/{repo}/issues/{PR_NUMBER}/comments
|
|
331
|
+
|
|
332
|
+
# Review comments (inline code comments)
|
|
333
|
+
gh api --paginate repos/{owner}/{repo}/pulls/{PR_NUMBER}/comments
|
|
334
|
+
|
|
335
|
+
# Review summaries (approve/request-changes/comment events)
|
|
336
|
+
gh api --paginate repos/{owner}/{repo}/pulls/{PR_NUMBER}/reviews
|
|
337
|
+
```
|
|
338
|
+
|
|
339
|
+
`--paginate` requests every REST page. These three calls are gate-allowed as
|
|
340
|
+
written; do not pipe any of them through `--jq`, `jq`, or another filter under
|
|
341
|
+
the PR_REVIEW gate — piped commands are blocked (fail-closed on `|`). To
|
|
342
|
+
separate bot/automated reviews (Copilot, Codex, CodeRabbit, etc.) from human
|
|
343
|
+
ones, apply the same predicate in context to the JSON already returned above
|
|
344
|
+
— `user.type == "Bot"` or a `user.login` match against
|
|
345
|
+
`bot|copilot|coderabbit|codex` (case-insensitive) — instead of re-fetching
|
|
346
|
+
with a shell-side `--jq` filter. `gh_evidence` with `target: "pr"` and
|
|
347
|
+
`fields: "comments"` is the sanctioned read-only tool path for the same PR
|
|
348
|
+
comment data when a tool call is preferred over a raw shell command. `gh pr
|
|
349
|
+
view --json comments,reviews` is convenience-only because those fields have
|
|
350
|
+
item caps; never use it as the authoritative "all signals" intake.
|
|
351
|
+
|
|
352
|
+
### Step 2 — Classify each comment
|
|
353
|
+
|
|
354
|
+
| Category | Action |
|
|
355
|
+
|----------|--------|
|
|
356
|
+
| **Human review with file:line evidence** | Add as candidate finding with `source: existing-review` — still needs reviewer validation |
|
|
357
|
+
| **Bot/automated finding with specific code reference** | Add as candidate finding with `source: bot-review` — high false-positive rate, treat as unverified |
|
|
358
|
+
| **General feedback / style preference** | Add as advisory obligation |
|
|
359
|
+
| **Resolved/outdated comment** | Skip — note in report under "Ingested Resolved Comments" |
|
|
360
|
+
| **Requested changes not yet addressed** | Add as HIGH-priority obligation |
|
|
361
|
+
|
|
362
|
+
### Step 3 — Merge into review pipeline
|
|
363
|
+
|
|
364
|
+
All ingested comments become candidate findings or obligations. They follow the
|
|
365
|
+
same Phase 3-8 pipeline as freshly discovered findings. Ingested findings are
|
|
366
|
+
NOT pre-confirmed — they still require independent reviewer validation per the
|
|
367
|
+
Anti-Self-Review Rule.
|
|
368
|
+
|
|
369
|
+
**Comment-ledger output:**
|
|
370
|
+
```
|
|
371
|
+
[INGESTED] | source | category | file:line (if applicable) | original_author | status: PENDING_VALIDATION / SKIPPED_OUTDATED / ADVISORY
|
|
372
|
+
```
|
|
373
|
+
|
|
374
|
+
### Anti-patterns
|
|
375
|
+
- ✗ Ignoring bot reviews because "bots produce false positives" — they also catch real issues
|
|
376
|
+
- ✗ Pre-confirming human review comments without independent validation — even senior reviewers make mistakes
|
|
377
|
+
- ✗ Skipping inline review comments and only reading the summary — inline comments contain the evidence
|
|
378
|
+
|
|
379
|
+
## Phase 0B: Mergeability and Branch-State Intake
|
|
380
|
+
|
|
381
|
+
Before investing effort in review lanes, verify the PR is mergeable and record
|
|
382
|
+
branch-state signals. `PR_REVIEW` remains read-only: do not resolve conflicts,
|
|
383
|
+
commit, push, rebase, merge, or reset from this mode. Instead, carry current
|
|
384
|
+
mergeability, stale-head, and branch-drift facts into the review ledger and the
|
|
385
|
+
feedback handoff artifact.
|
|
386
|
+
|
|
387
|
+
### Step 1 — Check merge state
|
|
388
|
+
|
|
389
|
+
The field names and values below are GitHub-specific examples. On another code
|
|
390
|
+
host, record the equivalent mergeability, conflict, required-check, base-drift,
|
|
391
|
+
and stale-head signals and preserve the same read-only behavior.
|
|
392
|
+
|
|
393
|
+
```bash
|
|
394
|
+
gh pr view <PR_NUMBER> --json mergeable,mergeStateStatus
|
|
395
|
+
```
|
|
396
|
+
|
|
397
|
+
The response has two independent fields. Handle each:
|
|
398
|
+
|
|
399
|
+
**`mergeable` field** — whether GitHub can compute mergeability:
|
|
400
|
+
| Value | Meaning | Action |
|
|
401
|
+
|-------|---------|--------|
|
|
402
|
+
| `MERGEABLE` | No conflicts detected | Proceed — check `mergeStateStatus` below |
|
|
403
|
+
| `CONFLICTING` | Merge conflicts exist | Record the blocker, keep the review read-only, and hand conflict resolution to `swarm-pr-feedback` |
|
|
404
|
+
| `UNKNOWN` | GitHub still computing | Wait 30s, re-check |
|
|
405
|
+
|
|
406
|
+
**`mergeStateStatus` field** — overall branch state:
|
|
407
|
+
| Value | Action |
|
|
408
|
+
|-------|--------|
|
|
409
|
+
| `CLEAN` | All checks pass, no conflicts — proceed to Phase 0 |
|
|
410
|
+
| `BEHIND` | Branch behind base — note in report; non-blocking if merge queue handles it |
|
|
411
|
+
| `DIRTY` | Merge conflicts exist — keep reviewing, but record the conflict as a first-class blocker in the ledger and handoff artifact |
|
|
412
|
+
| `BLOCKED` | External blocker (branch protection, failing required check) — investigate and record the blocker |
|
|
413
|
+
|
|
414
|
+
### Step 2 — Record conflicts and blockers (when CONFLICTING or DIRTY)
|
|
415
|
+
|
|
416
|
+
When the PR has merge conflicts:
|
|
417
|
+
|
|
418
|
+
1. **Determine the PR's base branch and verify the state**, as separate
|
|
419
|
+
standalone commands — never with `$(...)` command substitution, which the
|
|
420
|
+
PR_REVIEW gate blocks:
|
|
421
|
+
- Read the base ref: `gh pr view <PR_NUMBER> --json baseRefName` (or
|
|
422
|
+
`gh_evidence` with `target: "pr"`, `fields: "baseRefName"`).
|
|
423
|
+
- Fetch it by its literal value, substituted for `<base-ref>`:
|
|
424
|
+
`git fetch origin <base-ref>`.
|
|
425
|
+
- Re-check merge state: `gh pr view <PR_NUMBER> --json
|
|
426
|
+
mergeable,mergeStateStatus,baseRefName,headRefName`.
|
|
427
|
+
|
|
428
|
+
2. **Capture the affected scope without changing the branch:**
|
|
429
|
+
- List the files or subsystems implicated by the conflict if GitHub exposes them,
|
|
430
|
+
or note that the exact conflict set is still unknown.
|
|
431
|
+
- Identify whether the conflict appears mechanical (lockfile / generated output /
|
|
432
|
+
simple overlap) or semantic (logic changed on both sides). This is triage
|
|
433
|
+
signal for the follow-on feedback run, not permission to resolve it here.
|
|
434
|
+
|
|
435
|
+
3. **Record explicit next action for the handoff artifact:**
|
|
436
|
+
- `CONFLICT-### | mechanical | likely resolvable during pr-feedback`
|
|
437
|
+
- `CONFLICT-### | semantic | requires focused fix + validation during pr-feedback`
|
|
438
|
+
- `STALE-### | behind base by policy` when the branch is only stale, not conflicted
|
|
439
|
+
|
|
440
|
+
4. **Document in report:** List the branch-state facts, why they matter to the
|
|
441
|
+
review, and what `swarm-pr-feedback` must verify before it edits code.
|
|
442
|
+
|
|
443
|
+
### Conflict resolution anti-patterns
|
|
444
|
+
- ✗ Accepting "ours" or "theirs" for all conflicts without reading them
|
|
445
|
+
- ✗ Resolving semantic conflicts without understanding both sides
|
|
446
|
+
- ✗ Pushing resolution without running tests on the merged result
|
|
447
|
+
- ✗ Treating `PR_REVIEW` as the place to fix branch state — this mode stays read-only
|
|
448
|
+
|
|
449
|
+
## Phase 0B-bis: Pre-Handoff Parallel Work Snapshot
|
|
450
|
+
|
|
451
|
+
When the review surfaces findings that will likely need `swarm-pr-feedback`,
|
|
452
|
+
re-check for **parallel work** since the last fetch. The PR author, the bot
|
|
453
|
+
reviewer, or another swarm may have pushed commits while you were reviewing.
|
|
454
|
+
This is still read-only: capture the remote state so the handoff artifact starts
|
|
455
|
+
from the right branch facts.
|
|
456
|
+
|
|
457
|
+
### Step 1 — Compare remote state (read-only, no post-bind fetch)
|
|
458
|
+
|
|
459
|
+
Once the PR head is bound, the gate allows only the exact bound tracking
|
|
460
|
+
fetch (when one is armed), and a detached review HEAD has no tracking branch
|
|
461
|
+
to refetch against — `git fetch origin <pr-branch>` is blocked here. Compare
|
|
462
|
+
state through the read-only API instead, as one standalone command:
|
|
463
|
+
|
|
464
|
+
```bash
|
|
465
|
+
gh pr view <PR_NUMBER> --json headRefOid,commits
|
|
466
|
+
```
|
|
467
|
+
|
|
468
|
+
or the equivalent `gh_evidence` call with `target: "pr"` and
|
|
469
|
+
`fields: "headRefOid,commits"`. If the returned `headRefOid` differs from the
|
|
470
|
+
`pr_head_sha` bound at the start of this review, the remote has moved; the
|
|
471
|
+
`commits` field lists every commit's message and author to date, enough to
|
|
472
|
+
judge relevance to the pending findings. (The legitimate place to `git fetch`
|
|
473
|
+
is the pre-bind sequence under "Pre-flight git ref availability" above — this
|
|
474
|
+
step never repeats that fetch post-bind.)
|
|
475
|
+
|
|
476
|
+
### Step 2 — Evaluate new commits
|
|
477
|
+
|
|
478
|
+
For each new commit on the remote (identified by comparing `headRefOid` /
|
|
479
|
+
`commits` above against the SHA bound at the start of the review):
|
|
480
|
+
|
|
481
|
+
1. **Read the commit message from the `commits` field above.** For file
|
|
482
|
+
scope, use one standalone read-only call —
|
|
483
|
+
`gh api repos/{owner}/{repo}/commits/<sha>` — rather than a local
|
|
484
|
+
`git show`: the new commit's object is not fetched locally post-bind.
|
|
485
|
+
2. **Compare against the pending handoff scope:**
|
|
486
|
+
- Does the remote commit touch the same files as the validated findings?
|
|
487
|
+
- Does the remote commit appear to already address a finding you planned to
|
|
488
|
+
hand off?
|
|
489
|
+
- Does the remote commit introduce a new branch-state fact the handoff should
|
|
490
|
+
mention?
|
|
491
|
+
3. **Default stance: prefer the remote state as the next baseline.** When the
|
|
492
|
+
bundled copy is available (plugin runtimes), run the
|
|
493
|
+
`file:.swarm/bundled-skills/parallel-work-check/SKILL.md`
|
|
494
|
+
protocol for the formal decision template; otherwise apply the three
|
|
495
|
+
outcomes below directly. Record the outcome in the handoff artifact.
|
|
496
|
+
|
|
497
|
+
### Step 3 — Three outcomes
|
|
498
|
+
|
|
499
|
+
- **Parallel work supersedes:** Mark the older local checkout as stale in the
|
|
500
|
+
handoff artifact and tell `swarm-pr-feedback` to re-check out the current
|
|
501
|
+
remote head before editing.
|
|
502
|
+
- **Parallel work complements:** Carry both the validated findings and the new
|
|
503
|
+
remote commits into the handoff artifact so `swarm-pr-feedback` can verify the
|
|
504
|
+
combined state before patching.
|
|
505
|
+
- **Parallel work unrelated:** Note that the remote moved, but keep the same
|
|
506
|
+
validated finding set.
|
|
507
|
+
|
|
508
|
+
### Anti-patterns
|
|
509
|
+
|
|
510
|
+
- ✗ Pushing your fix without checking if the remote already fixed it — causes
|
|
511
|
+
duplicate work and may even fail the push if the commits conflict
|
|
512
|
+
- ✗ Force-pushing over parallel work because "I started this first" — the
|
|
513
|
+
parallel agent may have access to context you don't (different swarm
|
|
514
|
+
configuration, different model, different time budget)
|
|
515
|
+
- ✗ Blindly taking remote work without verifying it's actually better — the
|
|
516
|
+
parallel work may be incomplete or take a different approach that doesn't
|
|
517
|
+
match the original finding's intent
|
|
518
|
+
|
|
519
|
+
### Example: parallel swarm superseded local fix work
|
|
520
|
+
|
|
521
|
+
```
|
|
522
|
+
PARALLEL WORK CHECK (pre-fix):
|
|
523
|
+
- Branch: copilot/fix-legacy-hive-data-migration
|
|
524
|
+
- Local HEAD: 3c04997c fix: resolve PR #1238 review findings
|
|
525
|
+
- Remote HEAD: 79d7ec64 fix(knowledge-migrator): harden legacy migration loop
|
|
526
|
+
- Diverged: yes (remote is 2 commits ahead with more comprehensive fix)
|
|
527
|
+
- New commits on remote: 2
|
|
528
|
+
- Parallel swarm work detected: yes (different author)
|
|
529
|
+
- Decision: abandon-use-remote
|
|
530
|
+
- Rationale: Remote added 17 unit tests + try/catch error handling that
|
|
531
|
+
surpassed my planned batch-rewrite. Verified by re-running the test suite:
|
|
532
|
+
remote has 25/25 passing, my local plan would have produced 9/9.
|
|
533
|
+
```
|
|
534
|
+
|
|
535
|
+
---
|
|
536
|
+
|
|
537
|
+
# Default Review Workflow
|
|
538
|
+
|
|
539
|
+
## Phase 0: Context Pack and Review Signal Collection
|
|
540
|
+
|
|
541
|
+
Before launching explorers, build a compact `swarm-pr-review-context` in scratch or as a local artifact if file writes are allowed.
|
|
542
|
+
|
|
543
|
+
The context pack must include, when available:
|
|
544
|
+
|
|
545
|
+
```json
|
|
546
|
+
{
|
|
547
|
+
"scope": {
|
|
548
|
+
"base_ref": "...",
|
|
549
|
+
"head_ref": "...",
|
|
550
|
+
"commit_range": "...",
|
|
551
|
+
"changed_files": [],
|
|
552
|
+
"changed_hunks": [],
|
|
553
|
+
"public_api_changes": [],
|
|
554
|
+
"deleted_or_renamed_files": [],
|
|
555
|
+
"generated_files": []
|
|
556
|
+
},
|
|
557
|
+
"pr_metadata": {
|
|
558
|
+
"title": "...",
|
|
559
|
+
"body_claims": [],
|
|
560
|
+
"checkboxes": [],
|
|
561
|
+
"linked_issues": [],
|
|
562
|
+
"review_comments": [],
|
|
563
|
+
"commit_messages": []
|
|
564
|
+
},
|
|
565
|
+
"obligations": [],
|
|
566
|
+
"repo_graph": {
|
|
567
|
+
"source": ".swarm/repo-graph.json or fallback search",
|
|
568
|
+
"changed_symbols": [],
|
|
569
|
+
"callers": [],
|
|
570
|
+
"callees": [],
|
|
571
|
+
"imports": [],
|
|
572
|
+
"exports": [],
|
|
573
|
+
"sibling_implementations": []
|
|
574
|
+
},
|
|
575
|
+
"deterministic_signals": {
|
|
576
|
+
"ci": [],
|
|
577
|
+
"tests": [],
|
|
578
|
+
"coverage_delta": [],
|
|
579
|
+
"lint_typecheck_build": [],
|
|
580
|
+
"security_scanners": [],
|
|
581
|
+
"dependency_audit": [],
|
|
582
|
+
"secrets_scan": [],
|
|
583
|
+
"mutation_testing": []
|
|
584
|
+
},
|
|
585
|
+
"swarm_artifacts": {
|
|
586
|
+
"evidence_bundles": [],
|
|
587
|
+
"knowledge_hits": [],
|
|
588
|
+
"phase_state": [],
|
|
589
|
+
"metrics": []
|
|
590
|
+
},
|
|
591
|
+
"risk_triggers": []
|
|
592
|
+
}
|
|
593
|
+
```
|
|
594
|
+
|
|
595
|
+
### Context pack rules
|
|
596
|
+
|
|
597
|
+
- Diff-only review is allowed for quick orientation, but not enough to confirm nontrivial findings.
|
|
598
|
+
- For every changed production file, identify at least one caller, consumer, import path, route entrypoint, or reason none exists.
|
|
599
|
+
- If `.swarm/repo-graph.json` exists, use it to seed impact cones.
|
|
600
|
+
- If no repo graph exists, build a shallow impact cone using imports, exports, symbol search, route registration, CLI registration, or test references.
|
|
601
|
+
- Pull in relevant `.swarm/evidence/`, `.swarm/state`, `.swarm/knowledge`, or hive/project knowledge entries when present.
|
|
602
|
+
- Historical knowledge may guide candidate generation but cannot confirm a finding by itself.
|
|
603
|
+
- Mark stale, quarantined, or cross-project knowledge as advisory until independently verified in this repo.
|
|
604
|
+
|
|
605
|
+
---
|
|
606
|
+
|
|
607
|
+
## Review Finding Persistence
|
|
608
|
+
|
|
609
|
+
Do not rely on conversation context to preserve review findings. On Profile A,
|
|
610
|
+
use `write_pr_review_artifact` with `kind: "findings"`; the controller creates
|
|
611
|
+
and appends `.swarm/pr-review/<run_id>/findings.jsonl` without granting generic
|
|
612
|
+
write authority over `.swarm/`. On Profiles B/C, append the same records to a
|
|
613
|
+
`findings.jsonl` ledger file in your harness's session/task workspace (never
|
|
614
|
+
under `.swarm/`), with the review head SHA recorded at the top of the file.
|
|
615
|
+
|
|
616
|
+
Each persisted finding record must include at least:
|
|
617
|
+
|
|
618
|
+
```json
|
|
619
|
+
{"finding_id":"F-001","status":"PENDING","file_line":"src/file.ts:123","evidence":"quote, command output, lane id, or reviewer rationale","next_action":"route_to_reviewer"}
|
|
620
|
+
```
|
|
621
|
+
|
|
622
|
+
Minimum field contract:
|
|
623
|
+
|
|
624
|
+
- `finding_id`: stable ID from the candidate/reviewer/critic ledger.
|
|
625
|
+
- `status`: one of `PENDING`, `CONFIRMED`, `DISPROVED`, or `PRE_EXISTING`.
|
|
626
|
+
- `file_line`: exact `file:line` reference, or `N/A` with reason when the
|
|
627
|
+
finding is cross-file or artifact-only.
|
|
628
|
+
- `evidence`: compact source-backed proof, including lane/reviewer/critic IDs or
|
|
629
|
+
command output references when available.
|
|
630
|
+
- `next_action`: the next required action, such as `route_to_reviewer`,
|
|
631
|
+
`route_to_critic`, `report`, `suppress_with_reason`, or `handoff_to_feedback`.
|
|
632
|
+
|
|
633
|
+
Persist after every major validation boundary (Profile A via the controller
|
|
634
|
+
calls below; Profiles B/C by appending the same boundary-tagged records to the
|
|
635
|
+
ledger file):
|
|
636
|
+
|
|
637
|
+
1. **Post-explorer:** after Phase 3/4 candidate parsing and before reviewer
|
|
638
|
+
dispatch, call `write_pr_review_artifact` with `boundary: "post_explorer"`
|
|
639
|
+
and all candidates as `PENDING` with their lane provenance.
|
|
640
|
+
2. **Post-reviewer:** after Phase 6 reviewer validation, call the controller
|
|
641
|
+
with `boundary: "post_reviewer"` and update each reviewed
|
|
642
|
+
record to `CONFIRMED`, `DISPROVED`, `PRE_EXISTING`, or keep `PENDING` with a
|
|
643
|
+
concrete `next_action` if more evidence is required.
|
|
644
|
+
3. **Post-critic:** after Phase 8 critic challenge, call the controller with
|
|
645
|
+
`boundary: "post_critic"` and update final status,
|
|
646
|
+
severity/action notes in `evidence`, and final reporting or handoff action.
|
|
647
|
+
|
|
648
|
+
Resume/reload procedure:
|
|
649
|
+
|
|
650
|
+
1. Before continuing any compacted or resumed review, read the latest
|
|
651
|
+
`findings.jsonl` artifact and reconstruct the candidate/reviewer/critic
|
|
652
|
+
ledger from disk before dispatching more lanes.
|
|
653
|
+
2. If the artifact is missing but a review context says prior lanes ran, stop and
|
|
654
|
+
surface the missing artifact as a coverage gap instead of reclassifying from
|
|
655
|
+
memory.
|
|
656
|
+
3. Append new records rather than overwriting history unless the artifact format
|
|
657
|
+
explicitly tracks revisions; latest record for a `finding_id` wins during
|
|
658
|
+
reload.
|
|
659
|
+
|
|
660
|
+
---
|
|
661
|
+
|
|
662
|
+
## Phase 1: Intent Reconstruction / Obligation Extraction
|
|
663
|
+
|
|
664
|
+
Reconstruct what the PR is obligated to deliver before looking for bugs.
|
|
665
|
+
|
|
666
|
+
Use deterministic precedence, highest to lowest:
|
|
667
|
+
|
|
668
|
+
1. PR checkboxes and acceptance criteria,
|
|
669
|
+
2. linked issues / tickets,
|
|
670
|
+
3. explicit user request in the current conversation,
|
|
671
|
+
4. commit scopes and commit messages,
|
|
672
|
+
5. test names and test assertions,
|
|
673
|
+
6. interface diff / exported API changes,
|
|
674
|
+
7. changelog, README, migration, or docs edits,
|
|
675
|
+
8. LLM synthesis only when no higher-precedence source exists.
|
|
676
|
+
|
|
677
|
+
Output an obligation list:
|
|
678
|
+
|
|
679
|
+
```text
|
|
680
|
+
O-001 | source | claim | affected files/symbols | status: UNVERIFIED | evidence refs: []
|
|
681
|
+
```
|
|
682
|
+
|
|
683
|
+
For each obligation, record:
|
|
684
|
+
|
|
685
|
+
- source,
|
|
686
|
+
- exact claim,
|
|
687
|
+
- affected files or symbols,
|
|
688
|
+
- verification status: `UNVERIFIED → IN_PROGRESS → MET / PARTIALLY_MET / NOT_MET / UNVERIFIABLE`,
|
|
689
|
+
- linked finding ID when unmet,
|
|
690
|
+
- reason if unverifiable.
|
|
691
|
+
|
|
692
|
+
Tests are claims. A passing or added test does not prove the obligation unless the reviewer inspects the assertion strength and relevant code path.
|
|
693
|
+
|
|
694
|
+
### Quantitative claim verification
|
|
695
|
+
|
|
696
|
+
PR body numerical claims (test counts, coverage percentages, assertion counts, performance benchmarks) are obligations, not proof. For each quantitative claim:
|
|
697
|
+
|
|
698
|
+
1. Extract the claim and its source (PR body, comment, commit message).
|
|
699
|
+
2. Verify against actual tool output or CI artifacts when available.
|
|
700
|
+
3. If the claim cannot be independently verified, mark the obligation `UNVERIFIABLE` with reason.
|
|
701
|
+
4. If the claim is disproved by evidence, create a finding linking the discrepancy.
|
|
702
|
+
|
|
703
|
+
Common patterns to verify:
|
|
704
|
+
- "N tests pass" → count actual test results from CI logs or test runner output
|
|
705
|
+
- "N% coverage" → compare against coverage report
|
|
706
|
+
- "No regressions" → verify against test runner failure count
|
|
707
|
+
|
|
708
|
+
---
|
|
709
|
+
|
|
710
|
+
## Phase 2: Deterministic Signal Ingestion
|
|
711
|
+
|
|
712
|
+
Ingest deterministic signals as candidate generators. They are never final findings.
|
|
713
|
+
|
|
714
|
+
Use available local artifacts first. Run safe read-only or standard project validation commands only when appropriate for the environment.
|
|
715
|
+
|
|
716
|
+
Candidate signal sources include:
|
|
717
|
+
|
|
718
|
+
- CI failures and logs,
|
|
719
|
+
- test failures,
|
|
720
|
+
- coverage delta,
|
|
721
|
+
- lint/typecheck/build output,
|
|
722
|
+
- `git diff --check`,
|
|
723
|
+
- dependency audit output,
|
|
724
|
+
- lockfile diff,
|
|
725
|
+
- CodeQL alerts,
|
|
726
|
+
- Semgrep or SAST findings,
|
|
727
|
+
- secrets scan findings,
|
|
728
|
+
- license scan findings,
|
|
729
|
+
- mutation testing output,
|
|
730
|
+
- package manager warnings,
|
|
731
|
+
- generated schema diffs.
|
|
732
|
+
|
|
733
|
+
Record each signal as:
|
|
734
|
+
|
|
735
|
+
```text
|
|
736
|
+
[TOOL_CANDIDATE] | tool | severity | file:line | claim | raw_signal_summary | confidence
|
|
737
|
+
```
|
|
738
|
+
|
|
739
|
+
Tool candidate rules:
|
|
740
|
+
|
|
741
|
+
- Confirm reachability before reporting.
|
|
742
|
+
- Confirm PR-introducedness before reporting as a PR blocker.
|
|
743
|
+
- Confirm that a framework, schema, middleware, caller guard, or test isolation rule does not already mitigate it.
|
|
744
|
+
- Do not report scanner output verbatim without reviewer validation.
|
|
745
|
+
- Redact secrets; never paste raw credentials into the final output.
|
|
746
|
+
|
|
747
|
+
---
|
|
748
|
+
|
|
749
|
+
## Phase 3: Parallel Base Explorer Lanes
|
|
750
|
+
|
|
751
|
+
### Review depth tiers (size × risk)
|
|
752
|
+
|
|
753
|
+
Before dispatching, classify the PR into a depth tier from the context pack.
|
|
754
|
+
Record the tier and the active capability profile in the ledger and in the
|
|
755
|
+
final validation provenance. The tier scales how many subagents you spawn —
|
|
756
|
+
never which review dimensions or risk families get evaluated:
|
|
757
|
+
|
|
758
|
+
| Tier | Diff shape | Dispatch shape (Profiles B/C) |
|
|
759
|
+
|---|---|---|
|
|
760
|
+
| S | ≤ ~100 changed lines, ≤ 5 files, no risk triggers | Consolidate: 1–2 explorer lanes covering all six dimensions (B), or one candidate-generation pass (C); Phase 4 risk families fold into the same lanes as an explicit per-family checklist |
|
|
761
|
+
| M | ≤ ~1500 changed lines, or any risk trigger | Dedicated lanes for the triggered dimensions/families; consolidate the remaining thin dimensions into 1–2 lanes |
|
|
762
|
+
| L | > ~1500 changed lines, > ~50 files, multi-subsystem, or security-sensitive surface | Full fan-out: one lane per dimension (six) and per-family micro dispatch in Phase 4 |
|
|
763
|
+
|
|
764
|
+
Risk triggers (any one escalates to at least tier M, and the triggered
|
|
765
|
+
dimension/family always gets a dedicated lane at M and above):
|
|
766
|
+
auth/identity/sessions/permissions/secrets/cryptography; untrusted-input
|
|
767
|
+
parsing or new input/output boundaries; subprocess/shell/filesystem execution;
|
|
768
|
+
concurrency, state machines, retries, caching; dependency, lockfile, install,
|
|
769
|
+
CI, or release changes; public API, schema, config, or migration changes;
|
|
770
|
+
payments or PII handling; generated, vendored, or binary artifacts.
|
|
771
|
+
|
|
772
|
+
Scaling is one-directional: a larger tier or an active controller may demand
|
|
773
|
+
more lanes than the table; nothing — repository size, elapsed time, token
|
|
774
|
+
cost, or predicted simplicity — permits fewer lanes than the classified tier,
|
|
775
|
+
and no tier permits skipping a dimension or family. Under Profile A the
|
|
776
|
+
controller computes the tier itself from the bound `base_sha...pr_head_sha`
|
|
777
|
+
diff (`--numstat` totals; an uncomputable diff fails strict to tier L) and
|
|
778
|
+
mechanically enforces the matching floors on every base and micro batch. The
|
|
779
|
+
initial base wave and every micro batch keep the historical tier-L full
|
|
780
|
+
fan-out (six singleton base lanes on the initial base wave; one micro-lane
|
|
781
|
+
per family on every micro batch, not only the first), while tiers S and M
|
|
782
|
+
accept consolidated lanes that declare their complete `owned_workflow_lanes`
|
|
783
|
+
set — every dimension and family still owned exactly once and attested per
|
|
784
|
+
family. A tier-L base **retry** may consolidate dimensions that each have a
|
|
785
|
+
recorded, terminally-failed prior attempt and currently have no successful
|
|
786
|
+
source, subject to two lane floors: no single lane may own all six
|
|
787
|
+
dimensions, and — counted cumulatively across every recorded base batch, not
|
|
788
|
+
per batch, and including batches the capacity GC has since dropped — the six
|
|
789
|
+
dimensions must stay backed by at least four distinct lanes (each dimension
|
|
790
|
+
no consolidated lane claims counts as one, plus the FEWEST declared
|
|
791
|
+
consolidated lanes that suffice to cover the rest). That permits a small
|
|
792
|
+
consolidation as failure recovery and rejects re-doing the whole wave in two
|
|
793
|
+
or three lanes, whether the attempt is split across several batches or
|
|
794
|
+
disguised as overlapping or duplicate consolidations — declaring more lanes
|
|
795
|
+
than the cover needs buys no budget. A dimension that already has a
|
|
796
|
+
successful source, or whose prior attempt is still in flight, still requires
|
|
797
|
+
its own dedicated retry lane, and
|
|
798
|
+
a full six-lane singleton re-dispatch is always accepted. Risk triggers
|
|
799
|
+
remain caller-side escalation on every profile: dispatch MORE than the floor
|
|
800
|
+
whenever a trigger warrants it.
|
|
801
|
+
|
|
802
|
+
### Dispatch
|
|
803
|
+
|
|
804
|
+
Under Profile A, launch all base lanes with `dispatch_lanes_async`. Pass the six
|
|
805
|
+
lane specs together, set `mode: "swarm-pr-review:base"`, assign each lane its
|
|
806
|
+
exact `workflow_lane` identifier from the table below, set `max_concurrent` to
|
|
807
|
+
`6`, bind the batch with the exact current `pr_head_sha`, record the returned
|
|
808
|
+
`batch_id`, and pass the exact reviewed merge base and its base ref as
|
|
809
|
+
`base_sha` and `base_ref`. Use the REMOTE-TRACKING form for `base_ref`
|
|
810
|
+
(`origin/main`, not `main` or `refs/heads/main`) and compute `base_sha` against
|
|
811
|
+
that same ref, so the controller's recomputation matches yours. Every later base retry, micro, council, reviewer, and
|
|
812
|
+
critic dispatch repeats those same exact bindings. The controller recomputes
|
|
813
|
+
`git merge-base -- <base_ref> <pr_head_sha>`, rejects mismatches, and replaces
|
|
814
|
+
caller `scope` text with the complete verified `base_sha...pr_head_sha` PR diff;
|
|
815
|
+
caller scope is retained only as a non-authoritative focus hint. Continue only non-dependent architect
|
|
816
|
+
work: refine the obligation ledger, inspect PR metadata, prepare micro-lane
|
|
817
|
+
trigger checks, and run deterministic read-only local tools. The runtime rejects
|
|
818
|
+
partial, duplicate, mislabelled, or non-explorer base waves. Do not synthesize
|
|
819
|
+
findings from running lanes. Keep each lane `prompt` compact: send the shared
|
|
820
|
+
review context (PR diff, obligation ledger, scope) ONCE via the `common_prompt`
|
|
821
|
+
field, or have lanes read it from a file by absolute path, instead of inlining
|
|
822
|
+
the same large blob into all six prompts — oversized inline prompts produce
|
|
823
|
+
malformed or truncated tool-call JSON and force clumsy file workarounds.
|
|
824
|
+
|
|
825
|
+
All six dimensions must be covered on every PR — "small PR", "docs-only", and
|
|
826
|
+
"CI-only" change what each dimension examines, never whether it is evaluated.
|
|
827
|
+
Every dimension ends in its own `[CANDIDATE]` rows or a fully populated
|
|
828
|
+
per-dimension `[CLEAN]` attestation. Under Profile A at depth tier L this is an exact six-lane gate, not a soft target: the controller rejects an initial base wave with fewer than six singleton lanes, and the review is BLOCKED until the missing lanes are dispatched and settled; "time-saving" is not an exception. At tiers S and M the controller instead requires the initial wave's `owned_workflow_lanes` to partition all six dimensions exactly once across at least the tier's lane floor (S ≥ 1, M ≥ 3, `max_concurrent` equal to the lane count), and settlement demands per-dimension attestation from every consolidated lane — a lane that fails any owned dimension fails them all. Under Profiles B/C, the depth tier governs lane count the same way — a tier-S diff may cover the six dimensions in one or two consolidated lanes — while dimension coverage and per-dimension attestation remain mandatory.
|
|
829
|
+
|
|
830
|
+
Under Profile B, dispatch the same wave as parallel subagents through your
|
|
831
|
+
harness's subagent tool: one subagent per dimension by default, consolidated
|
|
832
|
+
per the depth tier for small diffs. Every lane prompt must carry the exact
|
|
833
|
+
`pr_head_sha`, the verified `base_sha...pr_head_sha` range, its assigned
|
|
834
|
+
`workflow_lane` identifier(s), and the explorer context contract below; append
|
|
835
|
+
every returned report to the findings ledger with its lane id and head SHA
|
|
836
|
+
before any reviewer dispatch. Under Profile C, run the same lanes as
|
|
837
|
+
sequential candidate-generation passes with the same per-lane ledger records.
|
|
838
|
+
The join barrier is universal: all base lanes settle before Phase 4 completes
|
|
839
|
+
or synthesis begins, whichever layer enforces it.
|
|
840
|
+
|
|
841
|
+
**Incremental collection (Profile A):** While base lanes are running, poll with `collect_lane_results` (without `wait` (or `wait: false`)) to check progress and process settled lanes as they complete — call `retrieve_lane_output` for full text when `output_ref` is present, then extract candidates via `parse_lane_candidates`, update the candidate ledger, validate output quality — while continuing independent architect work (obligation refinement, micro-lane trigger checks, local reads) between polls. Only use `wait: true` if lanes are still pending and no more independent work remains. Under Profile B, harvest each subagent report as it completes and update the ledger between arrivals; block on stragglers only when no independent work remains.
|
|
842
|
+
|
|
843
|
+
Inline `output` is delivered on the first poll that observes a lane settled; subsequent polls carry `output_omitted_repeat: true` with metadata and `output_ref`, and full text is retrieved via `retrieve_lane_output`.
|
|
844
|
+
|
|
845
|
+
Before Phase 4 or synthesis, all base lanes must be settled. `dispatch_lanes_async` accepts a maximum of 8 lanes per call; base lanes (6) and micro-lanes (Phase 4) are dispatched in separate calls by design. Do not let one lane's conclusions bias another lane.
|
|
846
|
+
|
|
847
|
+
**COVERAGE GATE — zero tolerance for unclosed gaps.** After `collect_lane_results`, verify every lane produced validated output. Two failure modes exist:
|
|
848
|
+
- **Mode A (empty output):** Lane returns 0 chars, `status: cancelled`, `output_digest` matches SHA-256 of empty string (`e3b0c442...b855`).
|
|
849
|
+
- **Mode B (intermediate reasoning only):** Lane reports `status: completed` with non-empty output, but the output is preliminary reasoning ("Now let me check...") with zero `[CANDIDATE]` rows and no parseable `[CLEAN] | workflow_lane | coverage_scope | evidence` attestation. The `output_digest` does NOT match the empty-string hash. `parse_lane_candidates` returns 0 candidates. This mode is MORE dangerous — the lane appears successful but produced no findings or clean proof.
|
|
850
|
+
|
|
851
|
+
For ANY lane that failed (either mode):
|
|
852
|
+
1. **Retry** (max 2 attempts) with materially different parameters — different session or prompt decomposition, while preserving the required structured async mode and exact head provenance.
|
|
853
|
+
2. If a base lane fails under Profile A, retry only the failed `workflow_lane` identifiers with `dispatch_lanes_async`, `mode: "swarm-pr-review:base"`, the same exact `pr_head_sha`, and explorer agents. The durable gate joins successful provenance across the initial wave and retry batches. While that controller is active, blocking `dispatch_lanes` and direct Task dispatch are not equivalent because they cannot satisfy the structured provenance gate. Under Profiles B/C, retry only the failed `workflow_lane` identifiers with a fresh subagent or pass, the same exact `pr_head_sha`, and a materially different prompt decomposition.
|
|
854
|
+
3. If no equivalent alternative can be verified, **STOP and surface the lane failure to the user as BLOCKED** with the lane id, scope, failure mode, retry attempts, and why equivalence could not be proven. Do not present partial findings, do not issue a review verdict, and do not synthesize from successful lanes. A low-quality partial review is worse than no review.
|
|
855
|
+
|
|
856
|
+
### Candidate extraction via parser
|
|
857
|
+
|
|
858
|
+
Under Profile A, after `collect_lane_results` returns for base lanes, process
|
|
859
|
+
each lane result that carries an `output_ref`. The orchestrator MUST use the
|
|
860
|
+
candidate parser rather than preview-text extraction:
|
|
861
|
+
|
|
862
|
+
1. For each singleton base `output_ref`, call `parse_lane_candidates` with
|
|
863
|
+
`output_ref`, `producer: "swarm-pr-review"`,
|
|
864
|
+
`expected_family: "base_explorer"`, and `expected_lane` set to the exact
|
|
865
|
+
`workflow_lane` declared at dispatch. For a consolidated tier-S/M lane, call
|
|
866
|
+
the parser once for each owned dimension with that dimension as
|
|
867
|
+
`expected_lane` and pass `expected_lanes` as the lane's complete
|
|
868
|
+
`owned_workflow_lanes` array on every call. The parser reads the full artifact
|
|
869
|
+
from disk (no preview truncation issue), rejects unowned rows, and returns
|
|
870
|
+
structured `ParseResultWithSidecar` records.
|
|
871
|
+
2. Filter the returned `candidates[]` by `producer: "swarm-pr-review"` plus the
|
|
872
|
+
exact `source_batch_id` and `source_lane_id` from the base dispatch. Treat a
|
|
873
|
+
family mismatch or parse error as a lane-output failure; family metadata is
|
|
874
|
+
not the acceptance boundary.
|
|
875
|
+
3. Group the filtered candidates into reviewer-sized chunks:
|
|
876
|
+
- by file area (group by the directory or module of the `file_line` field),
|
|
877
|
+
- by category (group by the `category` field),
|
|
878
|
+
- by count (target max 50 candidates per chunk; smaller chunks are fine).
|
|
879
|
+
4. Stage reviewer-sized chunks, but do not dispatch reviewers yet. Phase 4 must
|
|
880
|
+
complete trigger accounting and settle every launched micro-lane first.
|
|
881
|
+
|
|
882
|
+
If a lane has `output_degraded: true`, `transcript_incomplete: true`, or no usable `output_ref`, apply the COVERAGE GATE (Phase 3). Do not use blocking or direct-Task fallbacks while the controller is active, mark affected candidates UNVERIFIED to proceed, or infer candidate absence from a preview. Under Profiles B/C, a truncated, empty, or attestation-free subagent report is the same lane-output failure and takes the same COVERAGE GATE.
|
|
883
|
+
|
|
884
|
+
After candidate parsing and before reviewer dispatch, persist the post-explorer
|
|
885
|
+
candidate ledger using the Review Finding Persistence contract. This is the
|
|
886
|
+
durable recovery point for context compaction before Phase 6.
|
|
887
|
+
|
|
888
|
+
**Profiles B/C row convention:** without the parser, the `[CANDIDATE]` row
|
|
889
|
+
format is the extraction contract itself. Explorers emit the rows directly in
|
|
890
|
+
their reports (see the Explorer Prompt Template reference); the orchestrator
|
|
891
|
+
collects them verbatim, validates each row's field count and lane id, and
|
|
892
|
+
treats malformed rows — or output with neither `[CANDIDATE]` rows nor a fully
|
|
893
|
+
populated `[CLEAN]` attestation — as a lane-output failure under the COVERAGE
|
|
894
|
+
GATE. If the parser is unavailable under Profile A, the same row convention
|
|
895
|
+
applies as a fallback, but the orchestrator SHOULD use the parser as the
|
|
896
|
+
primary extraction mechanism.
|
|
897
|
+
|
|
898
|
+
**lane id uniqueness for parallel dispatches:** When re-dispatching failed or
|
|
899
|
+
re-running explorer lanes, every `dispatch_lanes_async` or `dispatch_lanes`
|
|
900
|
+
lane `id` MUST be unique within that dispatch batch and should include lane and
|
|
901
|
+
attempt suffixes (e.g., `pr_review_explore_lane1_attempt2`). Never reuse an id
|
|
902
|
+
in the same batch unless intentionally replacing that exact lane before dispatch.
|
|
903
|
+
|
|
904
|
+
Explorers optimize for recall. Over-reporting is expected. Explorers produce candidates only.
|
|
905
|
+
|
|
906
|
+
The six dimensions are a fixed **check-type** partition, not an area
|
|
907
|
+
partition: every PR needs all six review dimensions, and the lanes
|
|
908
|
+
deliberately overlap by file, each receiving the same diff (via
|
|
909
|
+
`common_prompt` under Profile A) and viewing it through a different lens. Six
|
|
910
|
+
dimensions are this workflow's high-assurance coverage floor, not a claim that
|
|
911
|
+
research proves a universal optimal agent count — the published evidence
|
|
912
|
+
favors complementary, distinct-lens reviewers over duplicated generalists, and
|
|
913
|
+
finding rates rise with diff size, which is why dispatch (not coverage)
|
|
914
|
+
follows the depth tier. Repository policy may add scrutiny but may never
|
|
915
|
+
reduce the six dimensions. Coverage is guaranteed by all six dimensions
|
|
916
|
+
reading the whole diff, so the disjoint-partition rule that governs area-split
|
|
917
|
+
fan-outs does not apply.
|
|
918
|
+
|
|
919
|
+
| `workflow_lane` | Focus | Required checks |
|
|
920
|
+
|---|---|---|
|
|
921
|
+
| `intent-architecture` | Intent, scope, architecture, and integration | obligation mapping, design fit, callers/consumers, sibling patterns, docs and claimed-vs-actual behavior |
|
|
922
|
+
| `correctness-state` | Functional correctness, data/state flow, edge cases, and failure paths | input domains, nullability, ordering, transactions, error behavior, rollback, backwards behavior |
|
|
923
|
+
| `tests-falsifiability` | Tests, test validity, regressions, and claimed validation | assertion strength, negative paths, isolation, fixtures, deterministic timing, missing proof |
|
|
924
|
+
| `security-trust` | Security, privacy, trust boundaries, unsafe inputs/sinks, and supply chain | authorization, injection, secrets, provenance, dependency risk, data exposure, abuse paths |
|
|
925
|
+
| `reliability-performance` | Reliability, concurrency, retries, resource bounds, and performance | races, retry semantics, timeouts, lifecycle, caching, algorithmic cost, operational failure modes |
|
|
926
|
+
| `compatibility-delivery` | API/schema/config compatibility, maintainability, build/deploy, docs, and release behavior | public contracts, migrations, runtime/platform support, packaging, CI, rollout and recovery guidance |
|
|
927
|
+
|
|
928
|
+
### Explorer context contract
|
|
929
|
+
|
|
930
|
+
Every explorer must inspect or explicitly mark unavailable:
|
|
931
|
+
|
|
932
|
+
1. the changed hunk,
|
|
933
|
+
2. at least one caller, consumer, or downstream impact-cone node,
|
|
934
|
+
3. at least one callee, dependency, or upstream assumption,
|
|
935
|
+
4. at least one sibling implementation or prior pattern,
|
|
936
|
+
5. the nearest relevant test or missing-test location,
|
|
937
|
+
6. deterministic signal entries mapped to its files/symbols,
|
|
938
|
+
7. relevant Swarm knowledge/evidence entries, if present.
|
|
939
|
+
8. the exact bound review range to analyze (`base_sha...pr_head_sha`),
|
|
940
|
+
|
|
941
|
+
### Explorer output format
|
|
942
|
+
|
|
943
|
+
Explorers emit structured candidate records. The parser reads the full lane
|
|
944
|
+
artifact and extracts these records. The canonical record shape is:
|
|
945
|
+
|
|
946
|
+
```text
|
|
947
|
+
[CANDIDATE] | candidate_id | lane | severity | category | file:line | claim | evidence_summary | impact_context | confidence
|
|
948
|
+
```
|
|
949
|
+
|
|
950
|
+
Profile A stores the full assistant transcript, so earlier unmarked progress
|
|
951
|
+
turns may precede the machine-readable section. The parser locates that section
|
|
952
|
+
at the first pipe-delimited line whose first field is exactly `[CANDIDATE]` and
|
|
953
|
+
ignores unmarked preamble, including incidental pipe-delimited text. That first
|
|
954
|
+
marker is authoritative and must be the exact canonical base or micro header;
|
|
955
|
+
a malformed marker or marker-prefixed data row without a header fails closed.
|
|
956
|
+
The Profile A controller coverage gate also refuses a missing marker. The pure
|
|
957
|
+
parser retains markerless positional fallback only for legacy callers outside
|
|
958
|
+
that controller trust boundary. Explorers should continue to make the canonical
|
|
959
|
+
header the first line of their final machine-readable response.
|
|
960
|
+
|
|
961
|
+
The confidence data value must be exactly LOW, MEDIUM, or HIGH.
|
|
962
|
+
|
|
963
|
+
Under Profile A the parser normalizes this into a structured `candidates[]`
|
|
964
|
+
array. On Profiles B/C — and as a Profile A fallback when the parser is
|
|
965
|
+
unavailable — the explorer emits the `[CANDIDATE]` row format directly in the
|
|
966
|
+
lane output as the extraction contract.
|
|
967
|
+
|
|
968
|
+
Explorers must not use `CONFIRMED`, `DISPROVED`, or `PRE_EXISTING`.
|
|
969
|
+
|
|
970
|
+
A base lane that finds no surviving candidates must emit exactly one fully
|
|
971
|
+
populated clean row:
|
|
972
|
+
|
|
973
|
+
```text
|
|
974
|
+
[CLEAN] | workflow_lane | coverage_scope | evidence
|
|
975
|
+
```
|
|
976
|
+
|
|
977
|
+
Header-only `[CLEAN]` markers, prose-only "clean" claims, or empty output do
|
|
978
|
+
not settle the lane.
|
|
979
|
+
|
|
980
|
+
---
|
|
981
|
+
|
|
982
|
+
## Phase 4: Mandatory Repository-Agnostic Micro-Lanes
|
|
983
|
+
|
|
984
|
+
After base lanes settle, inspect the exact diff/context pack to focus every row
|
|
985
|
+
in the micro-lane map and print a mandatory ledger with one row per map row:
|
|
986
|
+
|
|
987
|
+
```text
|
|
988
|
+
[TRIGGER-EVAL] | trigger_row | MATCHED/NOT_TRIGGERED | focus_evidence
|
|
989
|
+
```
|
|
990
|
+
|
|
991
|
+
Focus evidence must name the changed files, manifests, imports/symbols, semantic
|
|
992
|
+
signals, or explicit absence conditions. Use `MATCHED` when the exact diff has
|
|
993
|
+
an applicable surface and dispatch that family; use `NOT_TRIGGERED` only when
|
|
994
|
+
the row was evaluated and concrete absence evidence proves it inapplicable.
|
|
995
|
+
`unclassified-risk` is the always-`MATCHED` fallback. A `NOT_TRIGGERED` row is
|
|
996
|
+
not a waiver or a micro artifact and carries no source batch/lane provenance.
|
|
997
|
+
Repository identity, technology stack, PR size, elapsed time, or predicted risk
|
|
998
|
+
never justifies skipping a row.
|
|
999
|
+
|
|
1000
|
+
Every row in the map is a risk **family** that must be evaluated against the
|
|
1001
|
+
diff on every PR, in every repository. What scales with the depth tier is the
|
|
1002
|
+
dispatch shape — how many subagents carry that evaluation — never the
|
|
1003
|
+
evaluation itself. Each `MATCHED` family must end in its own attestation:
|
|
1004
|
+
`[CANDIDATE]` rows naming the family, or one fully populated per-family
|
|
1005
|
+
`[CLEAN]` row. `NOT_TRIGGERED` families end in the ledger with absence evidence
|
|
1006
|
+
and must not be dispatched.
|
|
1007
|
+
|
|
1008
|
+
**Profile A dispatch.** Launch the micro coverage with
|
|
1009
|
+
`dispatch_lanes_async` and `mode: "swarm-pr-review:micro"`. At depth tier L,
|
|
1010
|
+
dispatch one focused micro-lane for every `MATCHED` row, each lane's
|
|
1011
|
+
`workflow_lane` equal to its trigger ID; because the dispatcher accepts at
|
|
1012
|
+
most eight lanes per call, split large matched sets across bounded async
|
|
1013
|
+
batches. At tiers S and M,
|
|
1014
|
+
consolidated lanes may each own several families: set `workflow_lane` to one
|
|
1015
|
+
owned trigger ID and declare the complete `owned_workflow_lanes` set — every
|
|
1016
|
+
matched family owned exactly once across the dispatch, and every owned family
|
|
1017
|
+
attested in that lane's output, or the lane fails for all of them. Include
|
|
1018
|
+
the complete exact-set
|
|
1019
|
+
`trigger_evaluation` ledger and the same exact current `pr_head_sha` in every
|
|
1020
|
+
micro dispatch, in a separate batch from base lanes. The runtime rejects
|
|
1021
|
+
unrelated or duplicate micro-lanes within a batch, and final ledger persistence
|
|
1022
|
+
rejects any row whose completed owning-lane provenance is absent.
|
|
1023
|
+
Poll incrementally, then settle every launched lane. Persist
|
|
1024
|
+
the complete ledger with `write_pr_review_trigger_eval`; its rows use the stable
|
|
1025
|
+
trigger IDs below. Every `MATCHED` row includes its returned `source_batch_id`
|
|
1026
|
+
and `source_lane_id`; every `NOT_TRIGGERED` row must omit both fields. Missing,
|
|
1027
|
+
extra, duplicate, malformed, or incorrectly provenanced rows make persistence
|
|
1028
|
+
fail and Phase 4 BLOCKED. The tool atomically writes
|
|
1029
|
+
`.swarm/pr-review/<run_id>/trigger-eval.json`, separate from `findings.jsonl`;
|
|
1030
|
+
pass the exact reviewed merge-base as `base_sha`, the exact live base branch
|
|
1031
|
+
tip/ref used to compute it as `base_ref`, and the same `pr_head_sha` to the
|
|
1032
|
+
writer. The writer runs bounded `git merge-base -- <base_ref> <pr_head_sha>` and
|
|
1033
|
+
rejects any claimed `base_sha` that is not the exact result. It accepts only an
|
|
1034
|
+
exact eleven-row v2 receipt: `MATCHED` rows are backed by completed,
|
|
1035
|
+
non-degraded, exact-head artifacts from lanes that declared and attested their
|
|
1036
|
+
families; `NOT_TRIGGERED` rows are provenance-free. Counts are recomputed and
|
|
1037
|
+
must agree. It never uses keyword or path classification alone as absence
|
|
1038
|
+
evidence. Any head mismatch makes persistence fail. Historical unversioned and
|
|
1039
|
+
schema-v1 all-`MATCHED` receipts remain readable, but new writes are strict v2.
|
|
1040
|
+
Do not add trigger results to the finding-status enum.
|
|
1041
|
+
|
|
1042
|
+
**Profiles B/C dispatch.** Scale the lane shape to the depth tier while
|
|
1043
|
+
keeping all eleven family evaluations:
|
|
1044
|
+
|
|
1045
|
+
- Tier L: one focused lane per `MATCHED` family, mirroring Profile A.
|
|
1046
|
+
- Tier M: dispatch the `MATCHED` families across at least the controller's
|
|
1047
|
+
matched-set consolidation floor; `NOT_TRIGGERED` rows remain ledger-only.
|
|
1048
|
+
- Tier S: dispatch the `MATCHED` set in one or more consolidated micro lanes or
|
|
1049
|
+
sequentially separated passes; keep `NOT_TRIGGERED` rows ledger-only.
|
|
1050
|
+
|
|
1051
|
+
Whatever the dispatch shape: the ledger keeps one `[TRIGGER-EVAL]` row per
|
|
1052
|
+
family; each `MATCHED` row's focus evidence names the lane or pass that
|
|
1053
|
+
evaluated it and gets its own `[CANDIDATE]`/`[CLEAN]` attestation naming the
|
|
1054
|
+
family id; each `NOT_TRIGGERED` row records absence evidence without an
|
|
1055
|
+
artifact; and the completed ledger is persisted as `trigger-eval.json` in the
|
|
1056
|
+
session/task workspace before reviewer dispatch. A matched family with no
|
|
1057
|
+
attestation row is an unclosed coverage gap.
|
|
1058
|
+
|
|
1059
|
+
For each micro `output_ref` (Profile A), call `parse_lane_candidates` with
|
|
1060
|
+
`producer: "swarm-pr-review"`, `expected_family: "micro_lane"`, and
|
|
1061
|
+
`expected_micro_lane` set to the launch-micro-lane value from the
|
|
1062
|
+
provenance-linked trigger row. When the artifact came from a consolidated
|
|
1063
|
+
tier-S/M lane (its dispatch declared more than one `owned_workflow_lanes`
|
|
1064
|
+
entry), also pass `expected_micro_lanes` set to that lane's complete
|
|
1065
|
+
`owned_workflow_lanes` array — the same set already declared at micro
|
|
1066
|
+
dispatch time. Without it, the parser has no way to tell a sibling owned
|
|
1067
|
+
family's row from a genuinely out-of-scope one: every row belonging to the
|
|
1068
|
+
lane's other owned families is treated as a parse error instead of being
|
|
1069
|
+
skipped as out-of-scope, which can also invalidate that lane's own otherwise-valid
|
|
1070
|
+
`[CLEAN]` attestation for the family being extracted. Omit `expected_micro_lanes`
|
|
1071
|
+
only for a singleton (tier-L) lane. Accept a candidate only when its `producer`,
|
|
1072
|
+
`source_batch_id`, and `source_lane_id` match an allow-listed tuple from the
|
|
1073
|
+
original or retry micro dispatch and its `micro_lane` matches that trigger row;
|
|
1074
|
+
never filter acceptance by `row_format_family`. A zero-candidate artifact is
|
|
1075
|
+
clean only when the parser returns exactly one provenance-matching persisted
|
|
1076
|
+
`clean_attestation` whose `micro_lane` matches the trigger row, zero parse
|
|
1077
|
+
errors, zero malformed rows, and a complete, non-degraded source:
|
|
1078
|
+
|
|
1079
|
+
```text
|
|
1080
|
+
[CLEAN] | micro_lane | coverage_scope | evidence
|
|
1081
|
+
```
|
|
1082
|
+
|
|
1083
|
+
Header-only or malformed zero output is `UNATTESTED`; apply the COVERAGE GATE (Phase 3). Under Profile A, the structured async PR-workflow path is required to preserve `L1`, exact-head, batch, and workflow-lane provenance; the active controller rejects blocking and direct-Task substitutes, and Task-derived findings or CLEAN prose cannot satisfy Phase 4's controller ledger. Under Profiles B/C, acceptance is the row contract itself: accept a candidate or clean row only when its `micro_lane` field matches the trigger row it claims, and treat prose-only "clean" claims as `UNATTESTED`.
|
|
1084
|
+
|
|
1085
|
+
Each micro-lane receives:
|
|
1086
|
+
|
|
1087
|
+
- exact files and hunks in scope,
|
|
1088
|
+
- related obligations,
|
|
1089
|
+
- impact cone entries,
|
|
1090
|
+
- relevant deterministic signals,
|
|
1091
|
+
- related historical knowledge with quarantine/staleness status,
|
|
1092
|
+
- expected invariants,
|
|
1093
|
+
- structured candidate output — parser-extracted under Profile A; on Profiles
|
|
1094
|
+
B/C the micro-lane emits `[CANDIDATE]`/`[CLEAN]` rows directly as the
|
|
1095
|
+
extraction contract.
|
|
1096
|
+
|
|
1097
|
+
### Repository-agnostic mandatory micro-lane map
|
|
1098
|
+
|
|
1099
|
+
Every row is evaluated in every repository. Diff/context analysis determines
|
|
1100
|
+
whether it is `MATCHED` or `NOT_TRIGGERED`; paths or keywords alone are not
|
|
1101
|
+
sufficient absence evidence. Repository policy
|
|
1102
|
+
may require supplementary specialist review outside this canonical ledger, but
|
|
1103
|
+
supplementary work never replaces these portable rows. The `unclassified-risk`
|
|
1104
|
+
family is always `MATCHED` to cover novel failure modes and classification gaps.
|
|
1105
|
+
|
|
1106
|
+
> **Trigger-ID namespace — do not mix (issue #1931).** The `trigger_id` field
|
|
1107
|
+
> passed to `write_pr_review_trigger_eval` accepts **only** the 11 micro-lane
|
|
1108
|
+
> IDs in the table below. Three different namespaces appear in this skill and
|
|
1109
|
+
> they are NOT interchangeable:
|
|
1110
|
+
>
|
|
1111
|
+
> | Namespace | Example values | Used where? | Valid as `trigger_id`? |
|
|
1112
|
+
> | --- | --- | --- | --- |
|
|
1113
|
+
> | Micro-lane IDs (this table) | `auth-identity-secrets`, `untrusted-input-boundaries`, ... | `workflow_lane` of `swarm-pr-review:micro` dispatch; `trigger_id` of trigger-eval rows | **YES — only these** |
|
|
1114
|
+
> | Base-lane IDs | `intent-architecture`, `correctness-state`, `tests-falsifiability`, `security-trust`, `reliability-performance`, `compatibility-delivery` | `workflow_lane` of `swarm-pr-review:base` dispatch; validated by `enforcePrReviewBaseDimensions` | NO |
|
|
1115
|
+
> | Dispatch modes | `swarm-pr-review:base`, `swarm-pr-review:micro`, `swarm-pr-review:reviewer`, `swarm-pr-review:critic` | `mode` field of `dispatch_lanes_async` | NO |
|
|
1116
|
+
>
|
|
1117
|
+
> The writer rejects unknown trigger IDs with the list of valid IDs. Short
|
|
1118
|
+
> informal names (`correctness`, `security`, `deps`, `docs`, `tests`, `perf`)
|
|
1119
|
+
> sometimes appear in prose summaries; they are shorthand, not literal IDs.
|
|
1120
|
+
|
|
1121
|
+
| Trigger ID | Scope | Trigger in diff or context pack | Launch micro-lane | Invariants to check |
|
|
1122
|
+
|---|---|---|---|---|
|
|
1123
|
+
| `auth-identity-secrets` | universal | authentication, authorization, identity, sessions, permissions, secrets, cryptography | Identity and secret boundaries | least privilege, confused-deputy paths, credential lifecycle, cryptographic misuse, safe defaults |
|
|
1124
|
+
| `untrusted-input-boundaries` | universal | parsing, serialization, queries, templates/rendering, file or network input/output | Untrusted input and sink analysis | injection, traversal, SSRF, unsafe deserialization, output escaping, resource limits |
|
|
1125
|
+
| `subprocess-platform` | universal | subprocesses, shell commands, filesystem operations, OS/runtime-specific code | Subprocess and platform safety | array argv, bounded execution, path containment, portability, cleanup, non-interactive behavior |
|
|
1126
|
+
| `concurrency-state` | universal | queues, caches, retries, transactions, locks, state machines, async coordination | Concurrency and state transitions | races, atomicity, idempotency, retry accounting, rollback, stale state, bounded growth |
|
|
1127
|
+
| `dependencies-build-release` | universal | dependency manifests, lockfiles, installers, build scripts, CI, packaging, deployment | Dependency and delivery integrity | provenance, version/lock consistency, install safety, platform matrices, rollback and release completeness |
|
|
1128
|
+
| `api-schema-migrations` | universal | public API, wire/schema/config/storage formats, migrations, feature flags | Compatibility and migration safety | backward/forward compatibility, defaults, validation, mixed-version operation, recovery |
|
|
1129
|
+
| `test-infrastructure` | universal | tests, mocks, fixtures, harnesses, coverage, CI matrices | Test validity and isolation | meaningful assertions, contamination, determinism, negative paths, cross-platform proof, test theater |
|
|
1130
|
+
| `ui-accessibility-i18n` | universal | user interfaces, interaction flows, rendering, accessibility, localization | UI and human-interface quality | keyboard/screen-reader behavior, focus, error states, responsive behavior, locale-safe formatting |
|
|
1131
|
+
| `privacy-observability` | universal | telemetry, logs, analytics, traces, retention, diagnostics | Privacy and observability safety | minimization, redaction, consent, retention, stable metrics, non-gameable evidence |
|
|
1132
|
+
| `generated-provenance` | universal | generated, vendored, binary, model-produced, codegen or checked-in build artifacts | Generated artifact provenance | reproducibility, source linkage, tamper evidence, reviewable diffs, licensing and stale output |
|
|
1133
|
+
| `unclassified-risk` | universal | any changed artifact or behavior not confidently classified by the rows above | Unclassified high-risk fallback | full change-path review, hidden trust boundaries, novel failure modes, missing specialist classification |
|
|
1134
|
+
|
|
1135
|
+
Micro-lane output format:
|
|
1136
|
+
|
|
1137
|
+
```text
|
|
1138
|
+
[CANDIDATE] | candidate_id | micro_lane | severity | category | file:line | claim | invariant_violated | evidence_summary | confidence
|
|
1139
|
+
[CLEAN] | micro_lane | coverage_scope | evidence
|
|
1140
|
+
```
|
|
1141
|
+
|
|
1142
|
+
---
|
|
1143
|
+
|
|
1144
|
+
## Phase 5: Swarm-Native Verifier Routing
|
|
1145
|
+
|
|
1146
|
+
Use Swarm-native agents and artifacts when available. If exact agent names are unavailable, route the same task to the closest equivalent reviewer/critic role. On harnesses without the plugin, most `.swarm/` artifacts will not exist: mark those rows N/A in the validation provenance rather than fabricating them.
|
|
1147
|
+
|
|
1148
|
+
| Swarm verifier / artifact | When to use | Purpose |
|
|
1149
|
+
|---|---|---|
|
|
1150
|
+
| `critic_drift_verifier` | obligation-vs-code, docs-vs-code, phase/gate changes, schema/config changes | detect drift between stated behavior and actual implementation |
|
|
1151
|
+
| `critic_hallucination_verifier` | external APIs, package claims, URLs, CLI flags, GitHub behavior, model/tool names | verify claims against source or mark as unverified |
|
|
1152
|
+
| `curator_phase` | before exploration and after synthesis | retrieve relevant lessons; write back confirmed true positives / false positives |
|
|
1153
|
+
| `test_engineer` | confirmed/borderline correctness, security, state, schema, or config findings | propose or run falsification probes and regression tests |
|
|
1154
|
+
| `.swarm/repo-graph.json` | all nontrivial code changes | build impact cones and sibling-pattern checks |
|
|
1155
|
+
| `.swarm/evidence/` | schema, phase, state, council, and guardrail changes | verify evidence compatibility and serialized provenance |
|
|
1156
|
+
| Tool-returned `.swarm/evidence/` artifacts | after synthesis | record review quality only at paths actually returned by invoked evidence tools; never invent a metrics path |
|
|
1157
|
+
|
|
1158
|
+
Verifier output is advisory until incorporated by the independent reviewer or critic.
|
|
1159
|
+
|
|
1160
|
+
---
|
|
1161
|
+
|
|
1162
|
+
## Phase 6: Independent Reviewer Confirmation
|
|
1163
|
+
|
|
1164
|
+
**Reviewer-dispatch join barrier:** reviewer dispatch MUST NOT begin until the
|
|
1165
|
+
exact eleven-row micro-lane ledger is complete and persisted, every launched
|
|
1166
|
+
`MATCHED` micro lane is settled with its owned families attested, every
|
|
1167
|
+
`NOT_TRIGGERED` row has concrete absence evidence and no provenance, and every
|
|
1168
|
+
accepted micro result has parser-derived provenance (Profile A) or a valid
|
|
1169
|
+
CLEAN attestation.
|
|
1170
|
+
|
|
1171
|
+
Route candidates to reviewer subagents. The orchestrator routes candidates
|
|
1172
|
+
in bounded chunks produced by the candidate extraction in Phase 3-4. Each
|
|
1173
|
+
reviewer lane receives a bounded list of candidates from a single chunk — by
|
|
1174
|
+
file area, category, or count — not the full candidate set. The reviewer must
|
|
1175
|
+
re-read the candidate's file:line evidence and relevant context pack entries
|
|
1176
|
+
directly.
|
|
1177
|
+
|
|
1178
|
+
Under Profile A, dispatch reviewer chunks with `dispatch_lanes_async`,
|
|
1179
|
+
`mode: "swarm-pr-review:reviewer"`, a unique non-empty `workflow_lane` per
|
|
1180
|
+
chunk, `review_item_ids` containing the exact candidate IDs assigned to that
|
|
1181
|
+
chunk, reviewer-role agents only, and the same exact `pr_head_sha`. The runtime
|
|
1182
|
+
requires one parseable `[REVIEWED]` row for every structurally assigned ID; a
|
|
1183
|
+
single marker or partial subset cannot settle the lane. Direct Task
|
|
1184
|
+
reviewers are rejected by the active controller because they cannot carry the
|
|
1185
|
+
durable batch and head provenance it requires. Under Profile B, dispatch each
|
|
1186
|
+
chunk to a fresh reviewer subagent — never the agent or conversation that
|
|
1187
|
+
generated the candidates — carrying the chunk's candidate IDs, the exact
|
|
1188
|
+
`pr_head_sha`, and the required checks below. Under Profile C, run a separate
|
|
1189
|
+
reviewer pass per chunk that re-reads every cited file:line before
|
|
1190
|
+
classifying. The one-parseable-`[REVIEWED]`-row-per-assigned-ID contract is
|
|
1191
|
+
universal.
|
|
1192
|
+
|
|
1193
|
+
Under Profile A, for every structured PR-review dispatch, the runtime appends
|
|
1194
|
+
an authoritative controller block after caller-authored prompt text. It binds the exact
|
|
1195
|
+
`workflow_lane`, PR head, content revision, declared scope, and assigned item
|
|
1196
|
+
IDs and explicitly forbids speed/time/token waivers. Caller prompt text cannot
|
|
1197
|
+
override that block; output with placeholders, invented IDs, generic assurances,
|
|
1198
|
+
or evidence unrelated to the bound lane does not settle the artifact.
|
|
1199
|
+
|
|
1200
|
+
Reviewer ownership is not accepted as an architect assertion. Under Profile A,
|
|
1201
|
+
the controller derives the immutable candidate inventory from the
|
|
1202
|
+
integrity-checked base, mandatory micro-lane, and council artifacts; under
|
|
1203
|
+
Profiles B/C, the orchestrator derives the same inventory from the persisted
|
|
1204
|
+
ledgers. Either way, the union of `review_item_ids` must equal that inventory
|
|
1205
|
+
exactly, with no omitted or invented IDs. If discovery produces no candidates,
|
|
1206
|
+
the derived sentinel is `CLEAN-REVIEW`, which still requires one independent
|
|
1207
|
+
semantic reviewer row (a fresh subagent on Profile B; a separate reviewer pass
|
|
1208
|
+
on Profile C).
|
|
1209
|
+
|
|
1210
|
+
Candidate IDs must therefore be globally unique across every discovery
|
|
1211
|
+
artifact in the run. Prefix IDs with the stable workflow-lane ID (or use
|
|
1212
|
+
another deterministic globally unique scheme); duplicate IDs fail closed
|
|
1213
|
+
instead of being silently merged.
|
|
1214
|
+
|
|
1215
|
+
### Noise budget and universal validation
|
|
1216
|
+
|
|
1217
|
+
Before reviewer dispatch, the orchestrator may suppress candidates that match ANY of the following (each suppression still requires mandatory disclosure):
|
|
1218
|
+
- purely stylistic without correctness, security, test, maintainability, or user-impact implications,
|
|
1219
|
+
- exact duplicates of a candidate already queued for validation,
|
|
1220
|
+
- explorer-stated confidence=LOW with zero structural evidence (no file:line, no code path, no invariant reference).
|
|
1221
|
+
|
|
1222
|
+
Every suppressed candidate must appear in the final report under "Suppressed Candidates" with the reason. Suppression without disclosure is a hard rule violation.
|
|
1223
|
+
|
|
1224
|
+
**All remaining candidates — regardless of severity — must be routed to independent reviewer validation.** Severity alone does not determine validation eligibility; it determines routing priority. A LOW-severity candidate with file:line evidence and a specific code path gets the same reviewer attention as a HIGH-severity candidate.
|
|
1225
|
+
|
|
1226
|
+
Candidates not routed to reviewers must be listed as UNVERIFIED with reason in the validation provenance. Do not silently drop them.
|
|
1227
|
+
|
|
1228
|
+
### Reviewer required checks
|
|
1229
|
+
|
|
1230
|
+
For each candidate, the reviewer must determine:
|
|
1231
|
+
|
|
1232
|
+
- exact file:line evidence,
|
|
1233
|
+
- whether the issue is introduced by this PR or pre-existing,
|
|
1234
|
+
- reachability from realistic execution paths,
|
|
1235
|
+
- whether caller guards, schema validation, middleware, framework defaults, feature flags, or state-machine constraints mitigate it,
|
|
1236
|
+
- whether tests cover the negative path,
|
|
1237
|
+
- whether sibling files or docs must change together,
|
|
1238
|
+
- whether the severity is justified,
|
|
1239
|
+
- the smallest falsification probe that would prove or disprove it.
|
|
1240
|
+
|
|
1241
|
+
### Reviewer classifications
|
|
1242
|
+
|
|
1243
|
+
| Classification | Meaning |
|
|
1244
|
+
|---|---|
|
|
1245
|
+
| `CONFIRMED` | Evidence is real, reachable or structurally proven, and introduced or exposed by this PR |
|
|
1246
|
+
| `DISPROVED` | Candidate claim is incorrect, unreachable, mitigated, or based on a misunderstanding |
|
|
1247
|
+
| `UNVERIFIED` | Available evidence is insufficient to determine validity |
|
|
1248
|
+
| `PRE_EXISTING` | Issue exists on the base branch and is not materially worsened by this PR |
|
|
1249
|
+
|
|
1250
|
+
### Evidence classifications
|
|
1251
|
+
|
|
1252
|
+
| Type | Definition |
|
|
1253
|
+
|---|---|
|
|
1254
|
+
| `STRUCTURALLY_PROVEN` | File:line evidence directly demonstrates the bug or violated invariant |
|
|
1255
|
+
| `EXECUTION_PROVEN` | A test, trace, reproduction, or command demonstrates failure |
|
|
1256
|
+
| `STATIC_TRACE_PROVEN` | Static analysis plus reviewed path/context demonstrates reachability |
|
|
1257
|
+
| `PLAUSIBLE_BUT_UNVERIFIED` | Pattern suggests risk, but reachability or mitigation is unresolved |
|
|
1258
|
+
|
|
1259
|
+
Reviewer output format:
|
|
1260
|
+
|
|
1261
|
+
```text
|
|
1262
|
+
[REVIEWED] | candidate_id | classification | evidence_type | final_severity | introduced_by_pr: YES/NO/UNKNOWN | file:line | rationale | falsification_probe | reviewer_id
|
|
1263
|
+
```
|
|
1264
|
+
|
|
1265
|
+
For the mechanically derived `CLEAN-REVIEW` sentinel, use the same exact row
|
|
1266
|
+
with `DISPROVED | STRUCTURALLY_PROVEN | NONE | UNKNOWN | N/A` and concrete
|
|
1267
|
+
rationale/probe/reviewer fields; the sentinel means the reviewer independently
|
|
1268
|
+
found no surviving actionable candidate, not that reviewer validation was
|
|
1269
|
+
skipped.
|
|
1270
|
+
|
|
1271
|
+
Every reviewer response must end with one parseable `[REVIEWED]` row per
|
|
1272
|
+
assigned candidate. A malformed `[REVIEWED]` row is not a verdict: re-dispatch
|
|
1273
|
+
with the exact contract (max 2), then mark the reviewer dimension BLOCKED if no
|
|
1274
|
+
valid row returns.
|
|
1275
|
+
|
|
1276
|
+
`DISPROVED` findings must include the reason. `PRE_EXISTING` findings must include the base-branch evidence if available.
|
|
1277
|
+
|
|
1278
|
+
After reviewer lanes settle, persist the post-reviewer finding ledger before
|
|
1279
|
+
critic routing or synthesis. The artifact must preserve `CONFIRMED`,
|
|
1280
|
+
`DISPROVED`, `PRE_EXISTING`, and still-`PENDING` records with reviewer IDs and
|
|
1281
|
+
next actions.
|
|
1282
|
+
|
|
1283
|
+
---
|
|
1284
|
+
|
|
1285
|
+
## Phase 7: Falsification Probe Requirement
|
|
1286
|
+
|
|
1287
|
+
Each confirmed nontrivial finding must include at least one falsification artifact:
|
|
1288
|
+
|
|
1289
|
+
- runnable failing command,
|
|
1290
|
+
- proposed regression test,
|
|
1291
|
+
- mutation that current tests fail to kill,
|
|
1292
|
+
- static-analysis trace,
|
|
1293
|
+
- minimal execution path,
|
|
1294
|
+
- exact reason no runtime probe is available.
|
|
1295
|
+
|
|
1296
|
+
Nontrivial means any finding that affects correctness, security, state transitions, write authority, git safety, config, schema/evidence integrity, model/tool permissions, external fetches, persistence, or user-visible behavior.
|
|
1297
|
+
|
|
1298
|
+
A finding may still be reported without a runnable command if it is structurally proven, but the report must state why a runtime probe was not available.
|
|
1299
|
+
|
|
1300
|
+
---
|
|
1301
|
+
|
|
1302
|
+
## Phase 8: Critic Challenge
|
|
1303
|
+
|
|
1304
|
+
Route every reviewer-confirmed HIGH or CRITICAL finding to a critic. Also route borderline MEDIUM findings when they involve security, state machines, write authority, evidence integrity, model/tool permissions, git safety, or config ratchets.
|
|
1305
|
+
|
|
1306
|
+
The controller conservatively derives critic ownership from semantic reviewer
|
|
1307
|
+
rows: every reviewer-confirmed CRITICAL, HIGH, or MEDIUM item is mandatory
|
|
1308
|
+
critic inventory. This intentionally over-routes ordinary MEDIUM items because
|
|
1309
|
+
machine enforcement cannot safely infer every repository-specific trust
|
|
1310
|
+
boundary from prose. Completion is blocked until that exact derived inventory
|
|
1311
|
+
has valid critic rows.
|
|
1312
|
+
|
|
1313
|
+
Reviewer and critic settlement compose across batches, item by item: a phase
|
|
1314
|
+
settles once every review item in the current mechanically assigned inventory
|
|
1315
|
+
holds a successful verdict, whether that coverage comes from one batch or from
|
|
1316
|
+
several complementary partial retries. A later degraded, truncated, stale,
|
|
1317
|
+
wrong-identity, or malformed batch never supplies a verdict for the items it
|
|
1318
|
+
touches, but it does not discard verdicts other batches already supplied for
|
|
1319
|
+
different items.
|
|
1320
|
+
|
|
1321
|
+
When more than one successful batch covers the same item, the most recent
|
|
1322
|
+
successful batch wins that item. This conflict rule is one shared computation,
|
|
1323
|
+
so settlement and every downstream verdict use (candidate inventory, critic
|
|
1324
|
+
routing, final synthesis) never disagree about which claim is authoritative
|
|
1325
|
+
for an item. A batch contributes only when it was validated against the exact
|
|
1326
|
+
candidate inventory current at validation time; a batch recorded before that
|
|
1327
|
+
binding existed contributes only if it is wholly successful and its item set
|
|
1328
|
+
exactly matches the current inventory — the historical all-or-nothing rule,
|
|
1329
|
+
preserved unchanged for state that predates composition.
|
|
1330
|
+
|
|
1331
|
+
A critic claim is bound per item to the exact reviewer row it was validated
|
|
1332
|
+
against, not to the reviewer batch as a whole. A reviewer retry that
|
|
1333
|
+
reproduces a byte-identical row for an item retains that item's critic work;
|
|
1334
|
+
a reviewer row that changed at all — even one field — invalidates only that
|
|
1335
|
+
item's critic claim, not the whole critic wave. Critic batches recorded
|
|
1336
|
+
before per-item binding existed keep the old behavior: any newer reviewer
|
|
1337
|
+
batch invalidates them wholesale. Dispatch a fresh critic wave to cover
|
|
1338
|
+
whatever items composition leaves unclaimed; critic evidence can never
|
|
1339
|
+
predate the reviewer evidence it purports to challenge.
|
|
1340
|
+
|
|
1341
|
+
Settlement is item completeness, not lane completeness: a declared lane that
|
|
1342
|
+
never completes produces a diagnostic naming the abandoned lane, not an
|
|
1343
|
+
automatic block, as long as every item in the inventory already holds a
|
|
1344
|
+
successful verdict from some lane.
|
|
1345
|
+
|
|
1346
|
+
Under Profile A, dispatch critic chunks with `dispatch_lanes_async`,
|
|
1347
|
+
`mode: "swarm-pr-review:critic"`, a unique non-empty `workflow_lane` per
|
|
1348
|
+
chunk, `review_item_ids` containing the exact finding IDs assigned to that
|
|
1349
|
+
chunk, critic-role agents only, and the same exact `pr_head_sha`. The runtime
|
|
1350
|
+
requires one parseable `[CRITIC]` row for every structurally assigned ID and
|
|
1351
|
+
requires the reviewer phase to have settled — every item in the current
|
|
1352
|
+
inventory holding a successful reviewer verdict, composed across batches —
|
|
1353
|
+
before a critic wave.
|
|
1354
|
+
Under Profile B, dispatch each critic chunk to a fresh subagent that was
|
|
1355
|
+
neither the explorer nor the reviewer for those findings; under Profile C, run
|
|
1356
|
+
a separate critic pass. The one-parseable-`[CRITIC]`-row-per-assigned-ID
|
|
1357
|
+
contract and the reviewer-before-critic ordering are universal.
|
|
1358
|
+
|
|
1359
|
+
The critic must challenge:
|
|
1360
|
+
|
|
1361
|
+
- severity inflation,
|
|
1362
|
+
- weak or incomplete evidence,
|
|
1363
|
+
- missing mitigating context,
|
|
1364
|
+
- false reachability assumptions,
|
|
1365
|
+
- framework or middleware defaults,
|
|
1366
|
+
- schema validation gates,
|
|
1367
|
+
- state-machine constraints,
|
|
1368
|
+
- feature flags or dead code,
|
|
1369
|
+
- pre-existing status,
|
|
1370
|
+
- non-actionable or unsafe fix recommendations,
|
|
1371
|
+
- sibling-file gaps,
|
|
1372
|
+
- whether multiple comments should be grouped into one root cause.
|
|
1373
|
+
|
|
1374
|
+
Critic output format:
|
|
1375
|
+
|
|
1376
|
+
```text
|
|
1377
|
+
[CRITIC] | finding_id | UPHELD/DOWNGRADED/DISPROVED/NEEDS_MORE_EVIDENCE | final_severity | reason | required_report_change
|
|
1378
|
+
```
|
|
1379
|
+
|
|
1380
|
+
## Verdict row contract
|
|
1381
|
+
|
|
1382
|
+
The `[CRITIC]` row in the format above is **mandatory contract**, not advisory output. A critic response that does not end with that exact row format is treated as a planning preamble, not a verdict, and must be re-dispatched. Do not proceed past Phase 8 join barrier until each dispatched critic lane has produced a parseable `[CRITIC]` row.
|
|
1383
|
+
|
|
1384
|
+
**Re-dispatch trigger:** when a critic lane response is missing the verdict row, the orchestrator must automatically re-dispatch that lane with the explicit instruction: "Your final line MUST be exactly the Phase 8 contract row: `[CRITIC] | finding_id | UPHELD/DOWNGRADED/DISPROVED/NEEDS_MORE_EVIDENCE | final_severity | reason | required_report_change`. A response without that exact row will be treated as a planning message and re-dispatched." Do not synthesize findings from the planning preamble; only from the re-dispatched verdict.
|
|
1385
|
+
|
|
1386
|
+
`NEEDS_MORE_EVIDENCE` is deliberately non-terminal and never satisfies critic
|
|
1387
|
+
settlement. Re-dispatch a narrower critic/probe lane or report the dimension
|
|
1388
|
+
BLOCKED. Terminal critic rows are cross-field checked: `DISPROVED` requires
|
|
1389
|
+
`NONE`, `UPHELD` requires CRITICAL/HIGH/MEDIUM, and `DOWNGRADED` cannot remain
|
|
1390
|
+
CRITICAL.
|
|
1391
|
+
|
|
1392
|
+
**COVERAGE GATE alignment:** Critic lane failures apply the COVERAGE GATE (Phase 3) — under Profile A via `dispatch_lanes_async` with `mode: "swarm-pr-review:critic"` and the same exact `pr_head_sha`; under Profiles B/C via a fresh critic subagent or pass. Do NOT mark findings UNVERIFIED or continue past the gap. The orchestrator NEVER fabricates a critic verdict by parsing prose, by tolerating a planning preamble, by presenting partial findings, or by silently accepting reduced coverage.
|
|
1393
|
+
|
|
1394
|
+
Refuted findings become `DISPROVED` or `ADVISORY`, depending on critic rationale. Downgrades must be listed in the final validation provenance.
|
|
1395
|
+
|
|
1396
|
+
After critic lanes settle, persist the post-critic finding ledger before final
|
|
1397
|
+
synthesis. This artifact is the source of truth for resumed reporting and for
|
|
1398
|
+
any later `swarm-pr-feedback` handoff.
|
|
1399
|
+
|
|
1400
|
+
---
|
|
1401
|
+
|
|
1402
|
+
## Runtime-Aware False-Positive Guard Checklist
|
|
1403
|
+
|
|
1404
|
+
Before confirming any finding, the reviewer and critic must check all that apply:
|
|
1405
|
+
|
|
1406
|
+
- [ ] Schema validation gate: does schema validation reject malformed input before the flagged line?
|
|
1407
|
+
- [ ] Middleware interception: does middleware handle the request or command before the flagged path?
|
|
1408
|
+
- [ ] Framework default mitigation: does the framework inherently prevent this class of issue?
|
|
1409
|
+
- [ ] Caller context correctness: who invokes this code, and can untrusted input reach it?
|
|
1410
|
+
- [ ] Execution reachability: is the path reachable, or behind a feature flag, dead branch, build-only path, or commented-out code?
|
|
1411
|
+
- [ ] State-machine constraints: do ordering rules, locks, mutexes, phase gates, or transition guards prevent the state?
|
|
1412
|
+
- [ ] Permission boundary: does role/tool mapping prevent the operation?
|
|
1413
|
+
- [ ] Data lifetime: is the flagged state persisted, serialized, logged, or only transient?
|
|
1414
|
+
- [ ] Cross-platform behavior: does Windows/macOS/Linux path or shell behavior change the result?
|
|
1415
|
+
- [ ] Test environment mismatch: is the finding only true under a mock or fixture that cannot occur in production?
|
|
1416
|
+
|
|
1417
|
+
If a mitigation applies and was not accounted for, downgrade to `ADVISORY`, `UNVERIFIED`, or `DISPROVED`.
|
|
1418
|
+
|
|
1419
|
+
---
|
|
1420
|
+
|
|
1421
|
+
## Phase 9: Synthesis, Grouping, and Noise Budget
|
|
1422
|
+
|
|
1423
|
+
Before final output:
|
|
1424
|
+
|
|
1425
|
+
- group duplicate candidates by root cause,
|
|
1426
|
+
- report one finding per root cause,
|
|
1427
|
+
- attach all affected file:line references under that finding,
|
|
1428
|
+
- separate ship blockers from advisory notes,
|
|
1429
|
+
- suppress pure style/nit findings unless they indicate correctness, security, test, maintainability, or user-impact risk,
|
|
1430
|
+
- distinguish PR-introduced from pre-existing,
|
|
1431
|
+
- distinguish confirmed from plausible-but-unverified,
|
|
1432
|
+
- include disproved agent/tool claims,
|
|
1433
|
+
- keep final comments actionable.
|
|
1434
|
+
|
|
1435
|
+
### Finding ID format
|
|
1436
|
+
|
|
1437
|
+
```text
|
|
1438
|
+
F-001 | severity | category | root cause | affected file:line refs | reviewer | critic status
|
|
1439
|
+
```
|
|
1440
|
+
|
|
1441
|
+
### Suggested final grouping
|
|
1442
|
+
|
|
1443
|
+
1. Ship blockers,
|
|
1444
|
+
2. Important non-blockers,
|
|
1445
|
+
3. Test / coverage gaps,
|
|
1446
|
+
4. Pre-existing issues,
|
|
1447
|
+
5. Unverified plausible risks,
|
|
1448
|
+
6. Disproved candidates / false positives,
|
|
1449
|
+
7. Clean lane summary.
|
|
1450
|
+
|
|
1451
|
+
---
|
|
1452
|
+
|
|
1453
|
+
## Phase 10: Metrics and Knowledge Writeback
|
|
1454
|
+
|
|
1455
|
+
At the end of the review, include review quality metrics in the final report's
|
|
1456
|
+
validation provenance. Persist them only through an invoked evidence tool and
|
|
1457
|
+
record the exact `.swarm/evidence/` path returned by that tool; if no invoked
|
|
1458
|
+
tool supports metrics (including all of Profiles B/C), state `NOT PERSISTED —
|
|
1459
|
+
no metrics evidence writer` and keep the metrics block in the final report and
|
|
1460
|
+
session ledger rather than naming a nonexistent command or path.
|
|
1461
|
+
|
|
1462
|
+
Record:
|
|
1463
|
+
|
|
1464
|
+
- raw candidates by base lane,
|
|
1465
|
+
- raw candidates by micro-lane,
|
|
1466
|
+
- deterministic tool candidates,
|
|
1467
|
+
- reviewer-confirmed findings,
|
|
1468
|
+
- reviewer-disproved findings,
|
|
1469
|
+
- reviewer-unverified findings,
|
|
1470
|
+
- critic-upheld findings,
|
|
1471
|
+
- critic-downgraded findings,
|
|
1472
|
+
- critic-disproved findings,
|
|
1473
|
+
- final reported findings,
|
|
1474
|
+
- suppressed non-actionable candidates,
|
|
1475
|
+
- recurring false-positive patterns,
|
|
1476
|
+
- commands or probes used,
|
|
1477
|
+
- token/time cost if available,
|
|
1478
|
+
- accepted/fixed findings when known.
|
|
1479
|
+
|
|
1480
|
+
Knowledge writeback rules:
|
|
1481
|
+
|
|
1482
|
+
- Write back only validated true positives or validated false-positive patterns.
|
|
1483
|
+
- Include file patterns, invariant, evidence, and why it was confirmed/disproved.
|
|
1484
|
+
- Mark repo-specific lessons as project-tier unless there is strong evidence they generalize.
|
|
1485
|
+
- Never promote quarantined or unvalidated knowledge to hive-tier.
|
|
1486
|
+
- Never store secrets, private tokens, or raw sensitive logs.
|
|
1487
|
+
|
|
1488
|
+
---
|
|
1489
|
+
|
|
1490
|
+
## Phase 11: Post-Fix Re-verification
|
|
1491
|
+
|
|
1492
|
+
When the PR author pushes fixes after a review, perform a targeted re-verification before updating the verdict.
|
|
1493
|
+
|
|
1494
|
+
### Re-verification scope
|
|
1495
|
+
|
|
1496
|
+
Only re-verify findings the author claims to have fixed. Do not re-run the full review pipeline.
|
|
1497
|
+
|
|
1498
|
+
### Re-verification steps
|
|
1499
|
+
|
|
1500
|
+
1. For each finding the author claims fixed:
|
|
1501
|
+
a. Read the changed file(s) from the updated branch at the specific lines referenced in the original finding.
|
|
1502
|
+
b. Verify the fix addresses the root cause, not just the symptom.
|
|
1503
|
+
c. Check that the fix does not introduce a new issue in the same area.
|
|
1504
|
+
2. Run CI checks on the updated branch to confirm no regressions.
|
|
1505
|
+
3. For findings the author did not address, carry forward the original finding with unchanged status.
|
|
1506
|
+
|
|
1507
|
+
### Re-verification output
|
|
1508
|
+
|
|
1509
|
+
```
|
|
1510
|
+
[REVERIFIED] | finding_id | FIXED / PARTIALLY_FIXED / NOT_FIXED / NEW_ISSUE | evidence | updated_severity
|
|
1511
|
+
```
|
|
1512
|
+
|
|
1513
|
+
- `FIXED`: the root cause is resolved and no new issue introduced.
|
|
1514
|
+
- `PARTIALLY_FIXED`: the root cause is partially addressed or a residual concern remains.
|
|
1515
|
+
- `NOT_FIXED`: the root cause persists unchanged.
|
|
1516
|
+
- `NEW_ISSUE`: the fix introduced a new problem at the same location.
|
|
1517
|
+
|
|
1518
|
+
Update the verdict only after re-verifying all previously blocking findings.
|
|
1519
|
+
|
|
1520
|
+
---
|
|
1521
|
+
|
|
1522
|
+
For the full parser-based candidate extraction dry-run example, read `references/parser-dry-run.md`.
|
|
1523
|
+
|
|
1524
|
+
---
|
|
1525
|
+
|
|
1526
|
+
# Council Mode Workflow
|
|
1527
|
+
|
|
1528
|
+
Council mode is opt-in only and adversarial.
|
|
1529
|
+
|
|
1530
|
+
When triggered:
|
|
1531
|
+
|
|
1532
|
+
1. Build the same context pack as default mode.
|
|
1533
|
+
2. After the default base-dimension and risk-family coverage is complete, launch all supplementary council agents. Under Profile A, use one `dispatch_lanes_async` call with `mode: "swarm-pr-review:council"`, the same exact `pr_head_sha`, and one unique `workflow_lane` per council member; continue independent context preparation while they run, polling with `collect_lane_results` (without `wait`) to process settled agents incrementally, and use `wait: true` only when no independent work remains. All agents must be settled and their candidates added to the ledger before reviewer classification; under Profile A the runtime enforces this join barrier, and blocking, sequential, or direct-Task fallback is not equivalent to the structured council dispatch — bypassing the active controller is `BLOCKED`. Under Profile B, dispatch council members as parallel subagents with the same marker contract and settle them all before reviewer classification; under Profile C, run each council lens as a separate sequential pass.
|
|
1534
|
+
3. Each council agent assumes all work is wrong until code evidence proves otherwise.
|
|
1535
|
+
4. Each agent hunts within its lane only.
|
|
1536
|
+
5. Agents return the same mechanically parseable candidate contract as other discovery lanes: one `[CANDIDATE]` row per `EVIDENCE_FOUND` or `SUSPICIOUS` claim, or a fully populated `[CLEAN] | workflow_lane | coverage_scope | evidence` row when no candidate survives. Council prose without one of those markers does not settle the lane.
|
|
1537
|
+
6. Agents must not return `CONFIRMED`, `DISPROVED`, or final severity; candidate severity remains provisional until reviewer classification.
|
|
1538
|
+
7. The independent reviewer then classifies every council candidate as `CONFIRMED`, `DISPROVED`, `UNVERIFIED`, or `PRE_EXISTING`.
|
|
1539
|
+
8. Apply critic challenge to reviewer-confirmed HIGH/CRITICAL or borderline findings.
|
|
1540
|
+
9. Final synthesis distinguishes real blockers, real low-severity issues, accepted caveats, disproved council claims, and follow-up quality work.
|
|
1541
|
+
|
|
1542
|
+
Default council lanes:
|
|
1543
|
+
|
|
1544
|
+
- correctness and edge cases,
|
|
1545
|
+
- security and trust boundaries,
|
|
1546
|
+
- dependency and deployment safety,
|
|
1547
|
+
- docs and intent-vs-actual,
|
|
1548
|
+
- tests and falsifiability,
|
|
1549
|
+
- performance and architecture when risk justifies it.
|
|
1550
|
+
|
|
1551
|
+
Council prompt requirements:
|
|
1552
|
+
|
|
1553
|
+
- branch and commit range,
|
|
1554
|
+
- context pack summary,
|
|
1555
|
+
- files owned by that lane,
|
|
1556
|
+
- relevant impact cone,
|
|
1557
|
+
- explicit checklist,
|
|
1558
|
+
- strict output cap,
|
|
1559
|
+
- `EVIDENCE_FOUND / SUSPICIOUS / CLEAN` only,
|
|
1560
|
+
- file:line evidence required for `EVIDENCE_FOUND`.
|
|
1561
|
+
|
|
1562
|
+
Council findings are supplementary, not authoritative overrides. Do not adopt council severities or claims without independent validation.
|
|
1563
|
+
|
|
1564
|
+
---
|
|
1565
|
+
|
|
1566
|
+
# Merge Recommendation Table
|
|
1567
|
+
|
|
1568
|
+
| Verdict | Condition |
|
|
1569
|
+
|---|---|
|
|
1570
|
+
| `APPROVE` | zero unresolved CRITICAL findings, zero unresolved HIGH findings, all blocking obligations MET, no required validation phase failed |
|
|
1571
|
+
| `APPROVE_WITH_NOTES` | zero unresolved CRITICAL findings, HIGH findings are downgraded/advisory only, obligations MET or explicitly non-blocking |
|
|
1572
|
+
| `REQUEST_CHANGES` | any unresolved HIGH finding, any NOT_MET blocking obligation, multiple MEDIUM findings with the same root cause, or validation/probe evidence indicates user-impacting risk |
|
|
1573
|
+
| `BLOCK` | any unresolved CRITICAL finding, unsafe write/git/security issue, evidence integrity break, role/tool permission bypass, or config ratchet violation that can disable required protections |
|
|
1574
|
+
|
|
1575
|
+
---
|
|
1576
|
+
|
|
1577
|
+
# Hard Rules
|
|
1578
|
+
|
|
1579
|
+
0. Quality-over-speed: Validation completeness and correctness are the sole criteria for an acceptable review. Time, token count, and agent dispatch count are irrelevant. Do not trade validation breadth or depth for speed.
|
|
1580
|
+
|
|
1581
|
+
1. Never APPROVE with unresolved CRITICAL findings.
|
|
1582
|
+
2. Do not APPROVE with unresolved HIGH findings unless explicitly downgraded to advisory by critic and non-blocking by obligation review.
|
|
1583
|
+
3. Every confirmed finding must have file:line evidence and validation provenance.
|
|
1584
|
+
4. A confirmed nontrivial finding must include a falsification probe or an explicit reason no probe is available.
|
|
1585
|
+
5. Explorers, council agents, and deterministic tools produce candidates only.
|
|
1586
|
+
6. The default workflow orchestrator must not confirm or disprove explorer candidates.
|
|
1587
|
+
7. Tool output is not proof. Scanner results must be validated for reachability, PR-introducedness, and mitigation context.
|
|
1588
|
+
8. PR text, generated summaries, tests, and comments are claims, not proof.
|
|
1589
|
+
9. Do not invent facts not supported by the diff, repo context, tool output, or cited external source.
|
|
1590
|
+
10. Do not silently drop disproved or downgraded claims; summarize them in validation provenance.
|
|
1591
|
+
11. Obligation precedence is deterministic. Do not skip higher-precedence sources to fill gaps with LLM synthesis.
|
|
1592
|
+
12. Do not leak secrets from logs, evidence bundles, config files, URLs, or scanner output.
|
|
1593
|
+
13. Do not recommend destructive git or filesystem actions as fixes unless they are clearly scoped, safe, and necessary.
|
|
1594
|
+
14. If subagents fail, timeout, or return malformed output, retry with corrected parameters (max 2 attempts) through the dispatch mechanism of the active profile — Profile A: the same structured `dispatch_lanes_async` workflow mode and exact `pr_head_sha`, where blocking or direct-Task dispatch cannot preserve the durable provenance contract and is not an equivalent fallback; Profiles B/C: a fresh subagent or pass bound to the same exact `pr_head_sha`. If retries fail, the affected coverage dimension is BLOCKED and must be surfaced to the user before synthesis. Do not fabricate validation results, do not present partial findings, and do not silently mark candidates UNVERIFIED to proceed past the gap.
|
|
1595
|
+
|
|
1596
|
+
15. If context pack, repo graph, deterministic signals, or Swarm artifacts are unavailable, retry with alternative access paths. If a source that should exist on the active profile is still unavailable after retry, the affected coverage dimension is BLOCKED and must be surfaced to the user. A source that cannot exist on the active profile (for example `.swarm/` artifacts outside Profile A) is marked N/A in the validation provenance instead — N/A is disclosure, never a waiver of the dimensions and families that must still be covered. Do not proceed to synthesis with unclosed coverage gaps under a "best available evidence" rationale — the architect is not authorized to produce a degraded review.
|
|
1597
|
+
|
|
1598
|
+
---
|
|
1599
|
+
|
|
1600
|
+
# Pre-Synthesis Gate — Mandatory
|
|
1601
|
+
|
|
1602
|
+
Before writing the final output, print this checklist with filled values. Every blank field means the final output is invalid.
|
|
1603
|
+
|
|
1604
|
+
```text
|
|
1605
|
+
[VALIDATION] scope selected: ___
|
|
1606
|
+
[VALIDATION] capability profile (A/B/C) and depth tier (S/M/L): ___ / ___
|
|
1607
|
+
[VALIDATION] context pack built: YES/NO — ___
|
|
1608
|
+
[VALIDATION] obligation count: ___
|
|
1609
|
+
[VALIDATION] repo graph / impact cone source: ___
|
|
1610
|
+
[VALIDATION] deterministic signals ingested: ___
|
|
1611
|
+
[VALIDATION] lane dispatch mechanism: controller / native subagents / sequential passes — ___
|
|
1612
|
+
[VALIDATION] base dimensions covered with attestation: ___ / 6 (lanes dispatched: ___)
|
|
1613
|
+
[VALIDATION] base explorer lanes returned: ___ / ___
|
|
1614
|
+
[VALIDATION] micro risk families evaluated and attested: ___ / 11 OR BLOCKED — <missing rows> (micro lanes dispatched: ___)
|
|
1615
|
+
[VALIDATION] Swarm verifier routing used: ___
|
|
1616
|
+
[VALIDATION] raw candidates: ___
|
|
1617
|
+
[VALIDATION] tool candidates: ___
|
|
1618
|
+
[VALIDATION] reviewer lanes dispatched: ___
|
|
1619
|
+
[VALIDATION] reviewer lanes returned with parseable `[REVIEWED]` rows: ___ / ___
|
|
1620
|
+
[VALIDATION] findings confirmed by reviewer: ___
|
|
1621
|
+
[VALIDATION] findings rejected by reviewer as false positive: ___
|
|
1622
|
+
[VALIDATION] findings marked PRE_EXISTING: ___
|
|
1623
|
+
[VALIDATION] findings left UNVERIFIED: ___
|
|
1624
|
+
[VALIDATION] findings escalated to critic: ___
|
|
1625
|
+
[VALIDATION] critic dispatched: ___ OR "SKIPPED — no reviewer-confirmed HIGH/CRITICAL or borderline findings"
|
|
1626
|
+
[VALIDATION] critic returned: ___ OR "N/A"
|
|
1627
|
+
[VALIDATION] findings upheld by critic: ___
|
|
1628
|
+
[VALIDATION] findings downgraded by critic: ___
|
|
1629
|
+
[VALIDATION] findings disproved by critic: ___
|
|
1630
|
+
[VALIDATION] falsification probes included: ___
|
|
1631
|
+
[VALIDATION] grouped root-cause findings: ___
|
|
1632
|
+
[VALIDATION] metrics / knowledge writeback: ___
|
|
1633
|
+
[VALIDATION] all explorers verified to diff against PR branch, not HEAD: YES/NO
|
|
1634
|
+
[VALIDATION] noise-filter suppressed candidates: ___ (count, each with reason in final report)
|
|
1635
|
+
[VALIDATION] all non-suppressed candidates routed to reviewer: YES/NO
|
|
1636
|
+
```
|
|
1637
|
+
|
|
1638
|
+
If any reviewer lane lacks a parseable `[REVIEWED]` row after bounded
|
|
1639
|
+
re-dispatch, the reviewer dimension is BLOCKED. Do not infer or silently
|
|
1640
|
+
downgrade a verdict.
|
|
1641
|
+
|
|
1642
|
+
**COVERAGE GATE CONDITION:** If ANY validation dimension shows incomplete coverage (lanes that failed and were not closed by retry or verified equivalent alternative, CI that did not run, tools that were unavailable after retry), the Pre-Synthesis Gate FAILS — apply the COVERAGE GATE (Phase 3). Do not proceed to final output. Surface unclosed gaps with exact failing dimensions and retry/equivalence evidence.
|
|
1643
|
+
|
|
1644
|
+
---
|
|
1645
|
+
|
|
1646
|
+
# Final Output Format
|
|
1647
|
+
|
|
1648
|
+
Produce the final review in this order:
|
|
1649
|
+
|
|
1650
|
+
## PR intent
|
|
1651
|
+
|
|
1652
|
+
Summarize the obligations and user-visible intent.
|
|
1653
|
+
|
|
1654
|
+
## Implementation summary
|
|
1655
|
+
|
|
1656
|
+
Summarize what changed, including major files, public APIs, schemas, configs, tests, and Swarm artifacts.
|
|
1657
|
+
|
|
1658
|
+
## Intended vs actual mapping
|
|
1659
|
+
|
|
1660
|
+
| Obligation | Source | Actual evidence | Status | Linked finding |
|
|
1661
|
+
|---|---|---|---|---|
|
|
1662
|
+
|
|
1663
|
+
Use `MET`, `PARTIALLY_MET`, `NOT_MET`, or `UNVERIFIABLE`.
|
|
1664
|
+
|
|
1665
|
+
## Validation provenance
|
|
1666
|
+
|
|
1667
|
+
Include:
|
|
1668
|
+
|
|
1669
|
+
- context pack limitations,
|
|
1670
|
+
- explorer lanes launched and returned,
|
|
1671
|
+
- micro-lanes triggered,
|
|
1672
|
+
- deterministic signals ingested,
|
|
1673
|
+
- reviewer identity / role for each finding,
|
|
1674
|
+
- critic result for each escalated finding,
|
|
1675
|
+
- findings DISPROVED by reviewer with reason,
|
|
1676
|
+
- findings DOWNGRADED by critic with reason,
|
|
1677
|
+
- findings left UNVERIFIED with reason.
|
|
1678
|
+
|
|
1679
|
+
If zero findings, explicitly state:
|
|
1680
|
+
|
|
1681
|
+
```text
|
|
1682
|
+
No confirmed findings — all validated lanes CLEAN.
|
|
1683
|
+
```
|
|
1684
|
+
|
|
1685
|
+
Then provide a lane-by-lane clean summary.
|
|
1686
|
+
|
|
1687
|
+
## Confirmed findings
|
|
1688
|
+
|
|
1689
|
+
For each finding:
|
|
1690
|
+
|
|
1691
|
+
```text
|
|
1692
|
+
F-001 — Severity — Category — Root cause
|
|
1693
|
+
Files: path:line, path:line
|
|
1694
|
+
Status: CONFIRMED / critic status
|
|
1695
|
+
Evidence type: STRUCTURALLY_PROVEN / EXECUTION_PROVEN / STATIC_TRACE_PROVEN
|
|
1696
|
+
Why it matters:
|
|
1697
|
+
Validation:
|
|
1698
|
+
Falsification probe:
|
|
1699
|
+
Suggested fix:
|
|
1700
|
+
```
|
|
1701
|
+
|
|
1702
|
+
## Pre-existing findings
|
|
1703
|
+
|
|
1704
|
+
List separately from PR-introduced findings.
|
|
1705
|
+
|
|
1706
|
+
## Unverified but plausible risks
|
|
1707
|
+
|
|
1708
|
+
Only include if useful and clearly labeled as unverified.
|
|
1709
|
+
|
|
1710
|
+
## Test / coverage gaps
|
|
1711
|
+
|
|
1712
|
+
Focus on missing tests that would catch real risks, not generic coverage requests.
|
|
1713
|
+
|
|
1714
|
+
## Disproved candidates and false positives
|
|
1715
|
+
|
|
1716
|
+
List concise reasons for notable false positives from explorers, tools, council agents, or reviewers.
|
|
1717
|
+
|
|
1718
|
+
## Verdict
|
|
1719
|
+
|
|
1720
|
+
Use one of:
|
|
1721
|
+
|
|
1722
|
+
- `APPROVE`
|
|
1723
|
+
- `APPROVE_WITH_NOTES`
|
|
1724
|
+
- `REQUEST_CHANGES`
|
|
1725
|
+
- `BLOCK`
|
|
1726
|
+
|
|
1727
|
+
## Merge recommendation
|
|
1728
|
+
|
|
1729
|
+
Explain the recommendation in one short paragraph and list required actions before merge if applicable.
|
|
1730
|
+
|
|
1731
|
+
## Feedback handoff
|
|
1732
|
+
|
|
1733
|
+
When the review produced actionable validated findings or operational blockers,
|
|
1734
|
+
call `write_pr_review_artifact` with `kind: "handoff"` (Profile A). The controller writes
|
|
1735
|
+
`.swarm/pr-review/<run_id>/feedback-handoff.json` only when its finding IDs
|
|
1736
|
+
exactly match the latest confirmed `handoff_to_feedback` records. On Profiles
|
|
1737
|
+
B/C, write the same handoff content to the session/task workspace path
|
|
1738
|
+
described in "Handoff To PR Feedback" and reference that path in the
|
|
1739
|
+
continuation prompt. Include:
|
|
1740
|
+
|
|
1741
|
+
- the handoff artifact path,
|
|
1742
|
+
- the preserved finding IDs and provenance that `swarm-pr-feedback` must carry
|
|
1743
|
+
forward,
|
|
1744
|
+
- and an explicit question asking whether to continue into
|
|
1745
|
+
`swarm-pr-feedback`.
|
|
1746
|
+
|
|
1747
|
+
Use this exact continuation prompt format, substituting the exact path from
|
|
1748
|
+
whichever profile applies (`.swarm/pr-review/<run_id>/feedback-handoff.json`
|
|
1749
|
+
under Profile A, or the session/task workspace path under Profiles B/C — never
|
|
1750
|
+
mix the two):
|
|
1751
|
+
|
|
1752
|
+
```text
|
|
1753
|
+
/swarm pr-feedback <PR_URL> continue from <handoff_artifact_path>
|
|
1754
|
+
```
|
|
1755
|
+
|
|
1756
|
+
---
|
|
1757
|
+
|
|
1758
|
+
For reviewer, critic, and explorer prompt templates, read `references/prompt-templates.md`.
|
|
1759
|
+
|
|
1760
|
+
Under Profile A, after metrics and durable review artifacts are complete, but
|
|
1761
|
+
before emitting the user-facing final report, call `complete_pr_workflow` with
|
|
1762
|
+
mode `PR_REVIEW` and the same exact
|
|
1763
|
+
`pr_head_sha`. The tool refuses to clear the session gate while required base,
|
|
1764
|
+
trigger, declared reviewer/critic, or open-lane obligations remain incomplete.
|
|
1765
|
+
While the gate remains active, the runtime prepends a workflow-active banner
|
|
1766
|
+
to the first substantive text part of each architect message (the model's text
|
|
1767
|
+
is preserved below the banner; later parts of the same message, and blank
|
|
1768
|
+
parts, are left untouched) and re-wakes an idle parent session. A
|
|
1769
|
+
user interruption pauses every automatic wake path until a later explicit user
|
|
1770
|
+
turn settles; the durable gate remains available to continue or abort. Only
|
|
1771
|
+
emit the final report after the completion tool confirms that the gate cleared.
|
|
1772
|
+
|
|
1773
|
+
Under Profiles B/C, no mechanical response gate exists: the Pre-Synthesis Gate
|
|
1774
|
+
checklist is the completion gate. Emit the final report only after every
|
|
1775
|
+
checklist line is filled, every dimension and family is attested, and every
|
|
1776
|
+
BLOCKED item is surfaced.
|
|
1777
|
+
|
|
1778
|
+
## Aborting an unrecoverable review (Profile A)
|
|
1779
|
+
|
|
1780
|
+
The mechanical gate can leave the session stuck if the PR head cannot be
|
|
1781
|
+
fetched or checked out — for example when a compound `git fetch … && git
|
|
1782
|
+
checkout …` is repeatedly rejected as read-only shell syntax (the runtime
|
|
1783
|
+
requires each git intake command to be a single standalone command), when
|
|
1784
|
+
the PR ref is missing, or when the working tree is on the wrong branch and
|
|
1785
|
+
the merge-base bind can never verify. In that state the response gate
|
|
1786
|
+
suspends further auto-resumes for either of two independent reasons: a
|
|
1787
|
+
small number of consecutive unproductive wakes (the durable gate `revision`
|
|
1788
|
+
did not advance), or the total wake ceiling being reached (tier-scaled
|
|
1789
|
+
defaults S=12 / M=54 / L=102, overridable via the `totalWakeCeiling`
|
|
1790
|
+
option, and in-memory/per-process so the count resets on plugin reload and
|
|
1791
|
+
when the durable gate clears — but NOT across the PR_REVIEW → PR_FEEDBACK
|
|
1792
|
+
handoff, which keeps accumulating). Either suspension appends a
|
|
1793
|
+
`pr_workflow_wake_suspended` record to `.swarm/events.jsonl` naming the
|
|
1794
|
+
reason, both counters, the tier, and the ceiling in force — read that first
|
|
1795
|
+
when diagnosing why a review stopped resuming. Either way, the only exits
|
|
1796
|
+
are:
|
|
1797
|
+
|
|
1798
|
+
1. **Diagnose and retry the canonical standalone sequence.** Run
|
|
1799
|
+
`git fetch origin refs/pull/<N>/head`, verify
|
|
1800
|
+
`git cat-file -e <full_pr_head_sha>^{commit}`, then run
|
|
1801
|
+
`git switch --detach <full_pr_head_sha>`. Do not use `--track FETCH_HEAD`.
|
|
1802
|
+
Confirm `git rev-parse HEAD` equals the authoritative PR head, then recompute the exact
|
|
1803
|
+
merge base with `git merge-base -- <base_ref> <pr_head_sha>` (single
|
|
1804
|
+
command) and retry the `swarm-pr-review:base` dispatch with the exact
|
|
1805
|
+
`pr_head_sha`, `base_sha`, and `base_ref`.
|
|
1806
|
+
2. **Call `abort_pr_workflow`** with `mode: "PR_REVIEW"` and a one-line
|
|
1807
|
+
`reason` describing the blocker. The tool clears the durable gate state
|
|
1808
|
+
and stops the auto-resume loop. It refuses while PR workflow lanes are
|
|
1809
|
+
still in flight (collect their results with `collect_lane_results`
|
|
1810
|
+
first) and refuses once a PR_FEEDBACK workflow is armed for publication
|
|
1811
|
+
— in PR_REVIEW those refusals do not apply because there is no armed
|
|
1812
|
+
publication state. An audit event is appended to `.swarm/events.jsonl`.
|
|
1813
|
+
3. **Ask the user to run `/swarm abort-pr-workflow`** (a human-only
|
|
1814
|
+
restricted command; the agent cannot invoke it via `swarm_command`).
|
|
1815
|
+
This is the recovery path when the wake budget has suspended and the
|
|
1816
|
+
architect cannot make further tool progress.
|
|
1817
|
+
|
|
1818
|
+
Abort is a recovery tool, not a coverage shortcut. Use it only when the
|
|
1819
|
+
bind/checkout path is genuinely unreachable; never use it to skip a
|
|
1820
|
+
coverage obligation that is merely expensive or inconvenient.
|
|
1821
|
+
|
|
1822
|
+
On Profiles B/C there is no durable gate or auto-resume loop to clear: if the
|
|
1823
|
+
head bind is genuinely unreachable, report the blocker to the user and stop.
|