argus-reviewer-e2e 0.1.2 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -4,12 +4,20 @@
4
4
  <img src="docs/assets/social.png" alt="Argus — vision-model E2E testing" width="640" />
5
5
  </p>
6
6
 
7
+ <p align="center">
8
+ <a href="https://www.npmjs.com/package/argus-reviewer-e2e"><img src="https://img.shields.io/npm/v/argus-reviewer-e2e" alt="npm version" /></a>
9
+ <a href="https://github.com/duketopceo/Argus/actions/workflows/ci.yml"><img src="https://github.com/duketopceo/Argus/actions/workflows/ci.yml/badge.svg" alt="CI" /></a>
10
+ <a href="LICENSE"><img src="https://img.shields.io/badge/license-MIT-blue" alt="MIT license" /></a>
11
+ <a href="https://github.com/duketopceo/Argus/security/policy"><img src="https://img.shields.io/badge/security-policy-orange" alt="security policy" /></a>
12
+ </p>
13
+
7
14
  Open-source, self-hosted vision-model E2E testing — the hundred-eyed watcher for your UI. Bring your own `OPENROUTER_API_KEY`: record a flow once, fingerprint-cache every step, replay near-free, heal on UI drift, and get results as a check + comment on the GitHub PR.
8
15
 
9
16
  - **Vision-first**: a model looks at a screenshot and decides where to click — no selectors to write or maintain.
10
17
  - **Cache-first**: replay costs zero vision calls on an unchanged UI; heals re-spend only on drift and show up as reviewable cache diffs.
11
18
  - **Cost-explicit**: every call is metered from OpenRouter's per-call cost and rolled into a per-run dollar figure on the PR.
12
19
  - **Grounding specialist**: a `grounding_model` (e.g. a ui-tars-class model) can drive element location with its native coordinate output, verified against the DOM before any click executes.
20
+ - **Execution-backed review**: `code-review` findings carry CI evidence, and with the opt-in sandbox lane (`sandbox: { enabled: true }`) Argus authors a test probe for unexercised findings and runs it in a hardened, network-less Docker container — a finding that fails on head and passes on base is stamped **reproduced**, not just suspected.
13
21
 
14
22
  ```bash
15
23
  npm i -D argus-reviewer-e2e # or github:duketopceo/argus-reviewer
@@ -40,7 +48,10 @@ export default {
40
48
 
41
49
  The GitHub Action automatically sets `ARGUS_REVIEWER_TRACE` with the repository, PR number, commit, and run id, so every PR review is attributed in OpenRouter without extra config. You can also set `ARGUS_REVIEWER_TRACE` yourself (JSON object) to add more fields.
42
50
 
43
- Status: early development. See `action/` for the composite GitHub Action and `runner/` for self-hosted runner registration.
51
+ Status: early development. See `action/` for the composite GitHub Action,
52
+ `runner/` for self-hosted runner registration, `docs/quickstart.md` for
53
+ setup, `SECURITY.md` for the threat model, and `CONTRIBUTING.md` to hack
54
+ on it.
44
55
 
45
56
  ## File structure
46
57
 
@@ -52,10 +63,14 @@ argus-reviewer/
52
63
  ├── runner/ # Self-hosted runner registration docs + script
53
64
  │ ├── README.md
54
65
  │ └── register-runner.sh
66
+ ├── electron/ # Local observability dashboard (`npm run app`)
55
67
  ├── src/
56
68
  │ ├── api.ts # Test-facing `test`/`td` API + generated test file renderer
57
- │ ├── cli.ts # `record`, `run`, and `cache` commands
69
+ │ ├── cli.ts # record · run · code-review · delegate · cache · index · init
58
70
  │ ├── config.ts # `argus-reviewer.config.*` loader (legacy `vision-e2e.config.*` accepted)
71
+ │ ├── cache/
72
+ │ │ ├── fingerprint.ts # Per-step screenshot/a11y fingerprint + resolve
73
+ │ │ └── store.ts # Flow cache read/write
59
74
  │ ├── driver/
60
75
  │ │ ├── browser.ts # Playwright browser launch (chromium/firefox/webkit) + observation capture
61
76
  │ │ └── target.ts # Optional local dev-server target process
@@ -63,9 +78,19 @@ argus-reviewer/
63
78
  │ │ ├── actions.ts # Low-level page actions (click, type, scroll, …)
64
79
  │ │ ├── loop.ts # Vision model record/replay + healing loop
65
80
  │ │ └── prompts.ts # OpenRouter action/assertion prompts + JSON schemas
66
- │ ├── cache/
67
- │ │ ├── fingerprint.ts # Per-step screenshot/a11y fingerprint + resolve
68
- │ │ └── store.ts # Flow cache read/write
81
+ │ ├── evidence/
82
+ │ │ ├── ci.ts # PR metadata + CI check-run context for findings
83
+ │ │ ├── gate.ts # Fork-PR trust gate (argus-probe label bound to head SHA)
84
+ │ │ └── link.ts # Finding → evidence linkage + comment-safe sanitization
85
+ │ ├── executor/
86
+ │ │ ├── a0.ts # `a0 headless -p` delegation to a user's Agent Zero instance
87
+ │ │ └── sandbox.ts # Hardened Docker runner for generated probes
88
+ │ ├── index/ # Repo index, diff context, cache invalidation
89
+ │ ├── journal/ # Per-run structured journal entries
90
+ │ ├── probe/
91
+ │ │ ├── author.ts # Model-authored regression probe generation + validation
92
+ │ │ ├── harness.ts # vitest/jest/node:test detection + TAP classification
93
+ │ │ └── queue.ts # Head-vs-merge-base probe orchestration
69
94
  │ ├── report/
70
95
  │ │ ├── comment.ts # Markdown PR comment + commit-status rendering
71
96
  │ │ ├── junit.ts # JUnit XML output
package/action/action.yml CHANGED
@@ -42,6 +42,18 @@ inputs:
42
42
  description: Report output dir, passed to both run and code-review so the post step can find run.json/code-review.json regardless of config reportDir
43
43
  default: argus-reviewer-report
44
44
  required: false
45
+ sandbox:
46
+ description: Enable the B.2 probe lane — authored test probes for unexercised findings run in a hardened Docker container (no network, no secrets, read-only fs). Enable-only — 'false' does not override a config-enabled lane. Requires Docker on the runner; forced off on pull_request_target.
47
+ default: 'false'
48
+ required: false
49
+ max-comments:
50
+ description: Cap on inline review comments posted per run; overflow is summarized in the sticky. Overrides review.maxComments when set.
51
+ default: ''
52
+ required: false
53
+ run:
54
+ description: Run the browser-flow lane (`argus-reviewer run`). Set 'false' for code-review-only consumers — repos without a per-PR web target (CLIs, libraries, infra repos). Skips the Playwright install and run steps; the sticky comment and commit status then reflect code-review alone.
55
+ default: 'true'
56
+ required: false
45
57
  outputs:
46
58
  conclusion:
47
59
  description: 'Run conclusion: success, failure, or neutral'
@@ -61,6 +73,7 @@ runs:
61
73
  run: npm ci
62
74
 
63
75
  - name: Install Playwright browser
76
+ if: inputs.run != 'false'
64
77
  shell: bash
65
78
  working-directory: ${{ inputs.working-directory }}
66
79
  env:
@@ -89,6 +102,10 @@ runs:
89
102
  if: inputs.index == 'true'
90
103
  shell: bash
91
104
  working-directory: ${{ inputs.working-directory }}
105
+ env:
106
+ # Token only feeds trust resolution on issue_comment events —
107
+ # pull_request* events read fork status from the event payload.
108
+ GITHUB_TOKEN: ${{ github.token }}
92
109
  run: ${{ inputs.cli }} index
93
110
 
94
111
  - name: Run argus-reviewer code review
@@ -99,6 +116,8 @@ runs:
99
116
  env:
100
117
  OPENROUTER_API_KEY: ${{ inputs.openrouter-api-key }}
101
118
  GITHUB_TOKEN: ${{ github.token }}
119
+ ARGUS_SANDBOX: ${{ inputs.sandbox == 'true' && '1' || '' }}
120
+ ARGUS_MAX_COMMENTS: ${{ inputs.max-comments }}
102
121
  ARGUS_DEBUG: '1'
103
122
  ARGUS_REVIEWER_TRACE: >-
104
123
  {"repo":"${{ github.repository }}",
@@ -111,11 +130,13 @@ runs:
111
130
 
112
131
  - name: Run argus-reviewer
113
132
  id: run
133
+ if: inputs.run != 'false'
114
134
  shell: bash
115
135
  working-directory: ${{ inputs.working-directory }}
116
136
  continue-on-error: true
117
137
  env:
118
138
  OPENROUTER_API_KEY: ${{ inputs.openrouter-api-key }}
139
+ GITHUB_TOKEN: ${{ github.token }}
119
140
  ARGUS_DEBUG: '1'
120
141
  ARGUS_DIFF_BASE: ${{ inputs.diff-base }}
121
142
  ARGUS_BUDGET_USD: ${{ inputs.budget-usd }}
@@ -135,6 +156,7 @@ runs:
135
156
  OPENROUTER_API_KEY: ${{ inputs.openrouter-api-key }}
136
157
  VISION_E2E_WORKING_DIR: ${{ inputs.working-directory }}
137
158
  ARGUS_REPORT_DIR: ${{ inputs.report-dir }}
159
+ ARGUS_RUN_DISABLED: ${{ inputs.run == 'false' && '1' || '' }}
138
160
  with:
139
161
  github-token: ${{ github.token }}
140
162
  script: |
@@ -9,13 +9,23 @@ function formatUsd(n) {
9
9
  return `$${(n || 0).toFixed(6)}`
10
10
  }
11
11
 
12
+ /** Escape a report string for one markdown table cell. */
13
+ function cell(s) {
14
+ return String(s ?? '')
15
+ .replace(/\|/g, '\\|')
16
+ .replace(/[\r\n]+/g, ' ')
17
+ .slice(0, 200)
18
+ }
19
+
12
20
  function renderMissingKeyBody() {
13
21
  const lines = []
14
22
  lines.push(SENTINEL)
15
23
  lines.push('')
16
24
  lines.push('## argus-reviewer ⚪ skipped')
17
25
  lines.push('')
18
- lines.push('`OPENROUTER_API_KEY` is not configured. Add it as a repository or workflow secret to run argus-reviewer.')
26
+ lines.push(
27
+ '`OPENROUTER_API_KEY` is not configured. Add it as a repository or workflow secret to run argus-reviewer.',
28
+ )
19
29
  lines.push('')
20
30
  lines.push('This status is intentionally neutral, not a failure.')
21
31
  lines.push('')
@@ -28,14 +38,16 @@ function renderNoReportBody(reportDir, runUrl) {
28
38
  lines.push('')
29
39
  lines.push('## argus-reviewer ⚠️ no report')
30
40
  lines.push('')
31
- lines.push(`The run step produced no \`run.json\` under \`${reportDir}\`. The commit status fails closed — check the action logs before merging.`)
41
+ lines.push(
42
+ `The run step produced no \`run.json\` under \`${reportDir}\`. The commit status fails closed — check the action logs before merging.`,
43
+ )
32
44
  lines.push('')
33
45
  lines.push(`[View run](${runUrl})`)
34
46
  lines.push('')
35
47
  return lines.join('\n')
36
48
  }
37
49
 
38
- function renderBody(report, codeReview, runUrl, ok) {
50
+ function renderBody(report, codeReview, runUrl, ok, inlinePlan) {
39
51
  if (!report) return renderMissingKeyBody()
40
52
 
41
53
  const lines = []
@@ -68,7 +80,9 @@ function renderBody(report, codeReview, runUrl, ok) {
68
80
  lines.push(`- \`${path.basename(t.file)}\` — ${t.name}`)
69
81
  }
70
82
  lines.push('')
71
- lines.push(`**Risk:** ${ok ? 'Low — UI regression tests and code review passed; no heals or failures.' : 'High — investigate failures before merge.'}`)
83
+ lines.push(
84
+ `**Risk:** ${ok ? 'Low — UI regression tests and code review passed; no heals or failures.' : 'High — investigate failures before merge.'}`,
85
+ )
72
86
  lines.push('')
73
87
  if (Object.keys(trace).length > 0) {
74
88
  lines.push('**Trace**')
@@ -87,7 +101,9 @@ function renderBody(report, codeReview, runUrl, ok) {
87
101
  lines.push('| --- | --- | ---: | ---: | ---: | ---: |')
88
102
  for (const t of report.tests) {
89
103
  const result = t.ok ? '✅ pass' : '❌ fail'
90
- lines.push(`| ${t.name} | ${result} | ${t.visionCalls} | ${formatUsd(t.visionCostUsd)} | ${t.healEvents?.length ?? 0} | ${t.asserts?.length ?? 0} |`)
104
+ lines.push(
105
+ `| ${t.name} | ${result} | ${t.visionCalls} | ${formatUsd(t.visionCostUsd)} | ${t.healEvents?.length ?? 0} | ${t.asserts?.length ?? 0} |`,
106
+ )
91
107
  }
92
108
  lines.push('')
93
109
  lines.push('</details>')
@@ -173,14 +189,24 @@ function renderBody(report, codeReview, runUrl, ok) {
173
189
  lines.push('')
174
190
  lines.push('| Check | Status | Explanation |')
175
191
  lines.push('| --- | --- | --- |')
176
- lines.push(`| Tests | ${report.ok ? '✅ Passed' : '❌ Failed'} | ${report.totals.passed}/${report.totals.tests} tests passed |`)
177
- lines.push(`| Budget | ${report.totals.budgetExceeded ? '⚠️ Warning' : '✅ Passed'} | ${formatUsd(report.totals.visionCostUsd)} spent${budgetCap > 0 ? ` of ${formatUsd(budgetCap)}` : ''} |`)
178
- lines.push(`| Heal events | ${healCount === 0 ? '✅ Passed' : '⚠️ Warning'} | ${healCount} heal event${healCount === 1 ? '' : 's'} |`)
179
- lines.push(`| Assertions | ${assertFails === 0 ? '✅ Passed' : '❌ Failed'} | ${assertFails === 0 ? assertCount : `${assertFails} failed`} assertion${assertCount === 1 ? '' : 's'} |`)
192
+ lines.push(
193
+ `| Tests | ${report.ok ? '✅ Passed' : '❌ Failed'} | ${report.totals.passed}/${report.totals.tests} tests passed |`,
194
+ )
195
+ lines.push(
196
+ `| Budget | ${report.totals.budgetExceeded ? '⚠️ Warning' : '✅ Passed'} | ${formatUsd(report.totals.visionCostUsd)} spent${budgetCap > 0 ? ` of ${formatUsd(budgetCap)}` : ''} |`,
197
+ )
198
+ lines.push(
199
+ `| Heal events | ${healCount === 0 ? '✅ Passed' : '⚠️ Warning'} | ${healCount} heal event${healCount === 1 ? '' : 's'} |`,
200
+ )
201
+ lines.push(
202
+ `| Assertions | ${assertFails === 0 ? '✅ Passed' : '❌ Failed'} | ${assertFails === 0 ? assertCount : `${assertFails} failed`} assertion${assertCount === 1 ? '' : 's'} |`,
203
+ )
180
204
  lines.push(`| OpenRouter key | ✅ Passed | \`OPENROUTER_API_KEY\` configured |`)
181
205
  if (codeReview && !codeReview.skipped) {
182
206
  const codeStatus = codeReview.ok ? '✅ Passed' : '❌ Failed'
183
- lines.push(`| Code review | ${codeStatus} | ${codeReview.findings.length} findings (${codeReview.model}) |`)
207
+ lines.push(
208
+ `| Code review | ${codeStatus} | ${codeReview.findings.length} findings (${codeReview.model}) |`,
209
+ )
184
210
  } else {
185
211
  lines.push(`| Code review | ⚪ Skipped | ${codeReview?.summary ?? 'no report'} |`)
186
212
  }
@@ -188,29 +214,7 @@ function renderBody(report, codeReview, runUrl, ok) {
188
214
  lines.push('</details>')
189
215
  lines.push('')
190
216
 
191
- if (codeReview && !codeReview.skipped) {
192
- lines.push('<details>')
193
- lines.push('<summary>🧠 Code review</summary>')
194
- lines.push('')
195
- lines.push(`**Verdict:** ${codeReview.verdict} · ${codeReview.model} · ${codeReview.tokens}tok ${formatUsd(codeReview.visionCostUsd)}`)
196
- lines.push('')
197
- lines.push(codeReview.summary)
198
- lines.push('')
199
- if (codeReview.findings.length > 0) {
200
- const evidenceIcon = { exercised: '✅', corroborated: '🔴', not_exercised: '⚪', inconclusive: '❔' }
201
- lines.push('| File | Severity | Evidence | Finding |')
202
- lines.push('| --- | --- | --- | --- |')
203
- for (const f of codeReview.findings) {
204
- const ev = f.evidence
205
- ? `${evidenceIcon[f.evidence.status] ?? '❔'} ${f.evidence.detail}`
206
- : '—'
207
- lines.push(`| \`${f.file}\` | ${f.severity} | ${ev} | ${f.message} |`)
208
- }
209
- lines.push('')
210
- }
211
- lines.push('</details>')
212
- lines.push('')
213
- }
217
+ pushCodeReviewDetails(lines, codeReview, inlinePlan)
214
218
 
215
219
  lines.push('<details>')
216
220
  lines.push('<summary>✨ Actions</summary>')
@@ -228,50 +232,182 @@ function renderBody(report, codeReview, runUrl, ok) {
228
232
  return lines.join('\n')
229
233
  }
230
234
 
231
- async function main() {
232
- const pr = context.payload && context.payload.pull_request
233
- const owner = context.repo.owner
234
- const repo = context.repo.repo
235
- const hasKey = !!process.env.OPENROUTER_API_KEY
236
- const workDir = process.env.VISION_E2E_WORKING_DIR || ''
237
- const reportDir = path.resolve(
238
- process.env.GITHUB_WORKSPACE,
239
- workDir,
240
- process.env.ARGUS_REPORT_DIR || 'argus-reviewer-report',
235
+ // The 🧠 Code review details block — shared by the full body and the
236
+ // review-only body (run lane disabled).
237
+ function pushCodeReviewDetails(lines, codeReview, inlinePlan) {
238
+ if (!codeReview || codeReview.skipped) return
239
+ lines.push('<details>')
240
+ lines.push('<summary>🧠 Code review</summary>')
241
+ lines.push('')
242
+ lines.push(
243
+ `**Verdict:** ${codeReview.verdict} · ${codeReview.model} · ${codeReview.tokens}tok ${formatUsd(codeReview.visionCostUsd)}`,
241
244
  )
242
- const runUrl = `${process.env.GITHUB_SERVER_URL}/${owner}/${repo}/actions/runs/${process.env.GITHUB_RUN_ID}`
243
-
244
- let report
245
- let codeReview
246
- if (hasKey) {
247
- try {
248
- const raw = fs.readFileSync(path.join(reportDir, 'run.json'), 'utf8')
249
- report = JSON.parse(raw)
250
- } catch {
251
- report = undefined
245
+ // U7 triage record — Jev annotate/route signals, never the gate.
246
+ if (codeReview.triage) {
247
+ const t = codeReview.triage
248
+ if (t.unadjudicated === true) {
249
+ lines.push(` · 🧭 triage unadjudicated — Jev unavailable`)
250
+ } else {
251
+ lines.push(
252
+ ` · 🧭 triage: risk ${t.risk ?? '?'}/5` +
253
+ `${typeof t.needsDeepReview === 'number' ? ` · deep-review ${t.needsDeepReview.toFixed(2)}` : ''}` +
254
+ `${t.topRiskArea !== undefined ? ` · top area \`${cell(t.topRiskArea)}\`` : ''}` +
255
+ ` (${cell(t.mode)})`,
256
+ )
252
257
  }
253
- try {
254
- const raw = fs.readFileSync(path.join(reportDir, 'code-review.json'), 'utf8')
255
- codeReview = JSON.parse(raw)
256
- } catch {
257
- codeReview = undefined
258
+ }
259
+ if (Array.isArray(codeReview.probes) && codeReview.probes.length > 0) {
260
+ const reproduced = codeReview.probes.filter((p) => p.outcome === 'reproduced').length
261
+ lines.push(
262
+ ` · 🧪 ${codeReview.probes.length} probe${codeReview.probes.length === 1 ? '' : 's'} run, ${reproduced} reproduced`,
263
+ )
264
+ }
265
+ if (typeof codeReview.probeLaneSkipped === 'string') {
266
+ lines.push(` · 🧪 probe lane skipped — ${cell(codeReview.probeLaneSkipped)}`)
267
+ }
268
+ lines.push('')
269
+ lines.push(codeReview.summary)
270
+ lines.push('')
271
+ if (codeReview.findings.length > 0) {
272
+ const evidenceIcon = {
273
+ exercised: '✅',
274
+ corroborated: '🔴',
275
+ not_exercised: '⚪',
276
+ inconclusive: '❔',
277
+ reproduced: '🧪',
278
+ }
279
+ lines.push('| File | Severity | p | Category | Evidence | Finding |')
280
+ lines.push('| --- | --- | --- | --- | --- | --- |')
281
+ // Findings/evidence strings are model- and probe-emitted — sanitize
282
+ // for the markdown table and bound the section so an oversized report
283
+ // can't push the body past GitHub's 65536-char comment limit.
284
+ const MAX_FINDING_ROWS = 25
285
+ for (const f of codeReview.findings.slice(0, MAX_FINDING_ROWS)) {
286
+ const ev = f.evidence
287
+ ? `${evidenceIcon[f.evidence.status] ?? '❔'} ${cell(f.evidence.detail)}`
288
+ : '—'
289
+ // U8 — Jev P(true positive); unadjudicated findings render '—'.
290
+ const p = typeof f.p === 'number' ? f.p.toFixed(2) : '—'
291
+ lines.push(
292
+ `| \`${cell(f.file)}\` | ${cell(f.severity)} | ${p} | ${cell(f.category ?? '—')} | ${ev} | ${cell(f.message)} |`,
293
+ )
294
+ }
295
+ if (codeReview.findings.length > MAX_FINDING_ROWS) {
296
+ lines.push(
297
+ `| … | — | — | — | — | ${codeReview.findings.length - MAX_FINDING_ROWS} more findings in \`code-review.json\` |`,
298
+ )
299
+ }
300
+ lines.push('')
301
+ // Inline-comment cap note — dedup'd fresh findings the
302
+ // review.maxComments budget didn't post (TCA max_comments).
303
+ if (inlinePlan !== undefined && inlinePlan.dropped > 0) {
304
+ lines.push(
305
+ `*+${inlinePlan.dropped} inline-eligible finding(s) not posted — \`review.maxComments\` cap ${inlinePlan.cap}.*`,
306
+ )
307
+ lines.push('')
308
+ }
309
+ // Secrets-lane audit line — adjudicated/suppressed counts, never literals.
310
+ if (codeReview.secretsScan) {
311
+ if (typeof codeReview.secretsScan.skipped === 'string') {
312
+ lines.push(`*🔐 secrets scan skipped — ${cell(codeReview.secretsScan.skipped)}*`)
313
+ } else if (Array.isArray(codeReview.secretsScan.records)) {
314
+ const suppressed = codeReview.secretsScan.records.filter((r) => r.suppressed).length
315
+ const unadj = codeReview.secretsScan.records.filter((r) => !r.adjudicated).length
316
+ lines.push(
317
+ `*🔐 secrets scan: ${codeReview.secretsScan.records.length} candidate(s)` +
318
+ `${suppressed > 0 ? `, ${suppressed} adjudicated-suppressed` : ''}` +
319
+ `${unadj > 0 ? `, ${unadj} unadjudicated` : ''}` +
320
+ `${codeReview.secretsScan.overflow > 0 ? `, +${codeReview.secretsScan.overflow} over cap` : ''}.*`,
321
+ )
322
+ }
323
+ lines.push('')
258
324
  }
259
325
  }
326
+ // U8 adjudication audit — outside the findings guard so suppressed-
327
+ // only reviews still show what Jev removed. p values live on the
328
+ // findings table and in code-review.json records.
329
+ if (codeReview.findingAdjudication && Array.isArray(codeReview.findingAdjudication.records)) {
330
+ const fa = codeReview.findingAdjudication
331
+ if (fa.unadjudicated === true) {
332
+ lines.push('*🧮 adjudication: unadjudicated — Jev unavailable, nothing suppressed.*')
333
+ } else {
334
+ const suppressed = fa.records.filter((r) => r.suppressed).length
335
+ const unadj = fa.records.filter((r) => !r.adjudicated).length
336
+ lines.push(
337
+ `*🧮 adjudication: ${fa.records.length} finding(s) scored` +
338
+ `${suppressed > 0 ? `, ${suppressed} suppressed (nit/q)` : ''}` +
339
+ `${unadj > 0 ? `, ${unadj} unadjudicated` : ''}` +
340
+ `${fa.overflow > 0 ? `, +${fa.overflow} over cap` : ''}.*`,
341
+ )
342
+ }
343
+ lines.push('')
344
+ }
345
+ lines.push('</details>')
346
+ lines.push('')
347
+ }
260
348
 
261
- // Missing code-review.json after a continue-on-error step means the review
262
- // crashed, not that it skipped — an intentional skip writes ok+skipped.
263
- // Fail closed rather than reporting it as a clean skip.
264
- const codeReviewOk = codeReview != null && codeReview.ok === true
265
- const ok = (report?.ok === true) && codeReviewOk
266
- const conclusion = !hasKey ? 'neutral' : ok ? 'success' : 'failure'
267
- const body = !hasKey
268
- ? renderMissingKeyBody()
269
- : report === undefined
270
- ? renderNoReportBody(reportDir, runUrl)
271
- : renderBody(report, codeReview, runUrl, ok)
349
+ // Sticky body for `run: 'false'` consumers — no run.json exists by
350
+ // design, so the body and conclusion reflect code-review alone.
351
+ function renderReviewOnlyBody(codeReview, runUrl, ok, inlinePlan) {
352
+ const lines = []
353
+ lines.push(SENTINEL)
354
+ lines.push('')
355
+ lines.push(`## argus-reviewer ${ok ? '✅ PASS' : '❌ FAIL'}`)
356
+ lines.push('')
357
+ if (!codeReview) {
358
+ lines.push(
359
+ '**Summary:** code-review only (run lane disabled) — no `code-review.json` found. The review step crashed or produced no report; the commit status fails closed — check the action logs before merging.',
360
+ )
361
+ } else if (codeReview.skipped) {
362
+ lines.push(
363
+ `**Summary:** code-review only (run lane disabled) — review skipped: ${cell(codeReview.summary)}`,
364
+ )
365
+ } else {
366
+ lines.push(
367
+ `**Summary:** code review only (run lane disabled) · verdict **${codeReview.verdict}** · ` +
368
+ `${codeReview.findings.length} finding(s) · ${codeReview.model} · ` +
369
+ `${codeReview.tokens}tok ${formatUsd(codeReview.visionCostUsd)}`,
370
+ )
371
+ }
372
+ lines.push('')
373
+ lines.push('<details>')
374
+ lines.push('<summary>🚥 Pre-merge checks</summary>')
375
+ lines.push('')
376
+ lines.push('| Check | Status | Explanation |')
377
+ lines.push('| --- | --- | --- |')
378
+ lines.push('| OpenRouter key | ✅ Passed | `OPENROUTER_API_KEY` configured |')
379
+ if (codeReview && !codeReview.skipped) {
380
+ const codeStatus = codeReview.ok ? '✅ Passed' : '❌ Failed'
381
+ lines.push(
382
+ `| Code review | ${codeStatus} | ${codeReview.findings.length} findings (${codeReview.model}) |`,
383
+ )
384
+ } else {
385
+ lines.push(
386
+ `| Code review | ${codeReview ? '⚪ Skipped' : '❌ Failed'} | ${cell(codeReview?.summary ?? 'no report')} |`,
387
+ )
388
+ }
389
+ lines.push('')
390
+ lines.push('</details>')
391
+ lines.push('')
392
+ pushCodeReviewDetails(lines, codeReview, inlinePlan)
393
+ if (runUrl) lines.push(`[View run](${runUrl})`)
394
+ lines.push('')
395
+ lines.push('<details>')
396
+ lines.push('<summary>✨ Actions</summary>')
397
+ lines.push('')
398
+ lines.push('- [ ] Re-run argus-reviewer')
399
+ lines.push('')
400
+ lines.push('</details>')
401
+ lines.push('')
402
+ lines.push('---')
403
+ lines.push('')
404
+ lines.push('<sub>`argus-reviewer` — self-hosted, BYOK OpenRouter UI regression.</sub>')
405
+ lines.push('')
406
+ return lines.join('\n')
407
+ }
272
408
 
273
- async function postInlineComments(pr, codeReview) {
274
- if (!pr || !codeReview || codeReview.skipped || !codeReview.findings) return
409
+ async function planInlineComments(pr, codeReview) {
410
+ if (!pr || !codeReview || codeReview.skipped || !codeReview.findings) return undefined
275
411
  // Must match the severity vocabulary emitted by the code-review schema
276
412
  // (src/cli.ts): bug/risk are inline-worthy; nit/q stay in the sticky body.
277
413
  const inlineSeverities = ['bug', 'risk']
@@ -281,14 +417,25 @@ async function postInlineComments(pr, codeReview) {
281
417
  path: f.file,
282
418
  line: f.line,
283
419
  side: 'RIGHT',
284
- body: `**argus-reviewer ${f.severity}:** ${f.message}${f.evidence && f.evidence.status !== 'exercised' ? `\n\n*CI evidence: ${f.evidence.detail}*` : ''}`,
420
+ body: `**argus-reviewer ${f.severity}:** ${f.message}${
421
+ f.category ? ` \`${f.category}\`` : ''
422
+ }${
423
+ f.evidence && f.evidence.status === 'reproduced'
424
+ ? '\n\n*🧪 Reproduced by an Argus probe — fails on this PR head, clean on base. See workflow artifacts.*'
425
+ : f.evidence && f.evidence.status !== 'exercised'
426
+ ? `\n\n*CI evidence: ${f.evidence.detail}*`
427
+ : ''
428
+ }`,
285
429
  }))
286
- if (comments.length === 0) return
430
+ if (comments.length === 0) return { capped: [], dropped: 0, cap: 0 }
287
431
 
288
432
  // Re-runs on the same SHA must not duplicate inline comments — the sticky
289
433
  // body is upserted but review comments are not. Paginate fully (100/page)
290
434
  // and scope dedup to the current head: comments on older commits must not
291
- // suppress findings that still apply to this head.
435
+ // suppress findings that still apply to this head. The key is the body's
436
+ // first line (the finding itself) — trailing evidence notes like the
437
+ // `reproduced` upgrade change the body but must not re-post a duplicate.
438
+ const dedupKey = (path, line, body) => `${path}:${line}:${body.split('\n')[0]}`
292
439
  const posted = new Set()
293
440
  let page = 1
294
441
  for (;;) {
@@ -301,15 +448,24 @@ async function postInlineComments(pr, codeReview) {
301
448
  })
302
449
  for (const c of existing) {
303
450
  if (c.body && c.body.startsWith('**argus-reviewer') && c.commit_id === pr.head.sha) {
304
- posted.add(`${c.path}:${c.line}:${c.body}`)
451
+ posted.add(dedupKey(c.path, c.line, c.body))
305
452
  }
306
453
  }
307
454
  if (existing.length < 100) break
308
455
  page += 1
309
456
  }
310
- const fresh = comments.filter((c) => !posted.has(`${c.path}:${c.line}:${c.body}`))
311
- if (fresh.length === 0) return
457
+ const fresh = comments.filter((c) => !posted.has(dedupKey(c.path, c.line, c.body)))
458
+
459
+ // review.maxComments caps inline noise (TCA max_comments) — applied
460
+ // after dedup so already-posted comments don't eat the budget. A cap of
461
+ // 0 disables inline posting entirely (no empty review).
462
+ const cap = typeof codeReview.maxComments === 'number' ? codeReview.maxComments : 20
463
+ const capped = fresh.slice(0, cap)
464
+ return { capped, dropped: fresh.length - capped.length, cap }
465
+ }
312
466
 
467
+ async function postInlineComments(pr, plan) {
468
+ if (plan === undefined || plan.capped.length === 0) return
313
469
  // One batched review instead of N createReviewComment calls — avoids
314
470
  // secondary rate limits on large findings sets.
315
471
  try {
@@ -319,13 +475,65 @@ async function postInlineComments(pr, codeReview) {
319
475
  pull_number: pr.number,
320
476
  commit_id: pr.head.sha,
321
477
  event: 'COMMENT',
322
- comments: fresh,
478
+ comments: plan.capped,
323
479
  })
324
480
  } catch (e) {
325
481
  core.warning(`inline review failed: ${e.message}`)
326
482
  }
327
483
  }
328
484
 
485
+ async function main() {
486
+ const pr = context.payload && context.payload.pull_request
487
+ const owner = context.repo.owner
488
+ const repo = context.repo.repo
489
+ const hasKey = !!process.env.OPENROUTER_API_KEY
490
+ const workDir = process.env.VISION_E2E_WORKING_DIR || ''
491
+ const reportDir = path.resolve(
492
+ process.env.GITHUB_WORKSPACE,
493
+ workDir,
494
+ process.env.ARGUS_REPORT_DIR || 'argus-reviewer-report',
495
+ )
496
+ const runUrl = `${process.env.GITHUB_SERVER_URL}/${owner}/${repo}/actions/runs/${process.env.GITHUB_RUN_ID}`
497
+
498
+ let report
499
+ let codeReview
500
+ if (hasKey) {
501
+ try {
502
+ const raw = fs.readFileSync(path.join(reportDir, 'run.json'), 'utf8')
503
+ report = JSON.parse(raw)
504
+ } catch {
505
+ report = undefined
506
+ }
507
+ try {
508
+ const raw = fs.readFileSync(path.join(reportDir, 'code-review.json'), 'utf8')
509
+ codeReview = JSON.parse(raw)
510
+ } catch {
511
+ codeReview = undefined
512
+ }
513
+ }
514
+
515
+ // Missing code-review.json after a continue-on-error step means the review
516
+ // crashed, not that it skipped — an intentional skip writes ok+skipped.
517
+ // Fail closed rather than reporting it as a clean skip.
518
+ const codeReviewOk = codeReview != null && codeReview.ok === true
519
+ // run: 'false' consumers have no run.json by design — the conclusion then
520
+ // reflects the code-review verdict alone.
521
+ const runDisabled = process.env.ARGUS_RUN_DISABLED === '1'
522
+ const ok = (runDisabled || report?.ok === true) && codeReviewOk
523
+ const conclusion = !hasKey ? 'neutral' : ok ? 'success' : 'failure'
524
+ const inlinePlan = hasKey ? await planInlineComments(pr, codeReview) : undefined
525
+ const body = !hasKey
526
+ ? renderMissingKeyBody()
527
+ : runDisabled
528
+ ? renderReviewOnlyBody(codeReview, runUrl, ok, inlinePlan)
529
+ : report === undefined
530
+ ? renderNoReportBody(reportDir, runUrl)
531
+ : renderBody(report, codeReview, runUrl, ok, inlinePlan)
532
+
533
+ // Eligibility + dedup for inline comments, computed before the sticky
534
+ // body renders so the "+N not posted" note counts the *fresh* set — the
535
+ // cap is applied to fresh, not to raw findings (already-posted comments
536
+ // must not inflate the dropped count).
329
537
  if (pr) {
330
538
  const { data: comments } = await github.rest.issues.listComments({
331
539
  owner,
@@ -349,7 +557,7 @@ async function postInlineComments(pr, codeReview) {
349
557
  body,
350
558
  })
351
559
  }
352
- await postInlineComments(pr, codeReview)
560
+ await postInlineComments(pr, inlinePlan)
353
561
  }
354
562
 
355
563
  const sha = pr ? pr.head.sha : context.sha
package/dist/api.d.ts CHANGED
@@ -7,6 +7,7 @@ import { Config, defineConfig } from './config.js';
7
7
  import { ErrorRecord } from './journal/schema.js';
8
8
  import { Logger } from './log.js';
9
9
  export { defineConfig };
10
+ export { DecisionClient, DecisionError, JEV_DEFAULT_MODEL, type DecisionAnswer, type DecisionClientOptions, type DecisionErrorKind, type DecisionQuestion, } from './vision/decisions.js';
10
11
  /**
11
12
  * Test-facing API (R13). Test files are plain TypeScript using a `td` object:
12
13
  *
package/dist/api.js CHANGED
@@ -4,6 +4,7 @@ import { loadFlow, saveFlow } from './cache/store.js';
4
4
  import { Ledger } from './vision/ledger.js';
5
5
  import { defineConfig } from './config.js';
6
6
  export { defineConfig };
7
+ export { DecisionClient, DecisionError, JEV_DEFAULT_MODEL, } from './vision/decisions.js';
7
8
  const KEY_ALIASES = {
8
9
  tab: 'Tab',
9
10
  enter: 'Enter',