argus-reviewer-e2e 0.1.3 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -4,6 +4,13 @@
4
4
  <img src="docs/assets/social.png" alt="Argus — vision-model E2E testing" width="640" />
5
5
  </p>
6
6
 
7
+ <p align="center">
8
+ <a href="https://www.npmjs.com/package/argus-reviewer-e2e"><img src="https://img.shields.io/npm/v/argus-reviewer-e2e" alt="npm version" /></a>
9
+ <a href="https://github.com/duketopceo/Argus/actions/workflows/ci.yml"><img src="https://github.com/duketopceo/Argus/actions/workflows/ci.yml/badge.svg" alt="CI" /></a>
10
+ <a href="LICENSE"><img src="https://img.shields.io/badge/license-MIT-blue" alt="MIT license" /></a>
11
+ <a href="https://github.com/duketopceo/Argus/security/policy"><img src="https://img.shields.io/badge/security-policy-orange" alt="security policy" /></a>
12
+ </p>
13
+
7
14
  Open-source, self-hosted vision-model E2E testing — the hundred-eyed watcher for your UI. Bring your own `OPENROUTER_API_KEY`: record a flow once, fingerprint-cache every step, replay near-free, heal on UI drift, and get results as a check + comment on the GitHub PR.
8
15
 
9
16
  - **Vision-first**: a model looks at a screenshot and decides where to click — no selectors to write or maintain.
@@ -41,7 +48,10 @@ export default {
41
48
 
42
49
  The GitHub Action automatically sets `ARGUS_REVIEWER_TRACE` with the repository, PR number, commit, and run id, so every PR review is attributed in OpenRouter without extra config. You can also set `ARGUS_REVIEWER_TRACE` yourself (JSON object) to add more fields.
43
50
 
44
- Status: early development. See `action/` for the composite GitHub Action and `runner/` for self-hosted runner registration.
51
+ Status: early development. See `action/` for the composite GitHub Action,
52
+ `runner/` for self-hosted runner registration, `docs/quickstart.md` for
53
+ setup, `SECURITY.md` for the threat model, and `CONTRIBUTING.md` to hack
54
+ on it.
45
55
 
46
56
  ## File structure
47
57
 
@@ -53,10 +63,14 @@ argus-reviewer/
53
63
  ├── runner/ # Self-hosted runner registration docs + script
54
64
  │ ├── README.md
55
65
  │ └── register-runner.sh
66
+ ├── electron/ # Local observability dashboard (`npm run app`)
56
67
  ├── src/
57
68
  │ ├── api.ts # Test-facing `test`/`td` API + generated test file renderer
58
- │ ├── cli.ts # `record`, `run`, and `cache` commands
69
+ │ ├── cli.ts # record · run · code-review · delegate · cache · index · init
59
70
  │ ├── config.ts # `argus-reviewer.config.*` loader (legacy `vision-e2e.config.*` accepted)
71
+ │ ├── cache/
72
+ │ │ ├── fingerprint.ts # Per-step screenshot/a11y fingerprint + resolve
73
+ │ │ └── store.ts # Flow cache read/write
60
74
  │ ├── driver/
61
75
  │ │ ├── browser.ts # Playwright browser launch (chromium/firefox/webkit) + observation capture
62
76
  │ │ └── target.ts # Optional local dev-server target process
@@ -64,9 +78,19 @@ argus-reviewer/
64
78
  │ │ ├── actions.ts # Low-level page actions (click, type, scroll, …)
65
79
  │ │ ├── loop.ts # Vision model record/replay + healing loop
66
80
  │ │ └── prompts.ts # OpenRouter action/assertion prompts + JSON schemas
67
- │ ├── cache/
68
- │ │ ├── fingerprint.ts # Per-step screenshot/a11y fingerprint + resolve
69
- │ │ └── store.ts # Flow cache read/write
81
+ │ ├── evidence/
82
+ │ │ ├── ci.ts # PR metadata + CI check-run context for findings
83
+ │ │ ├── gate.ts # Fork-PR trust gate (argus-probe label bound to head SHA)
84
+ │ │ └── link.ts # Finding → evidence linkage + comment-safe sanitization
85
+ │ ├── executor/
86
+ │ │ ├── a0.ts # `a0 headless -p` delegation to a user's Agent Zero instance
87
+ │ │ └── sandbox.ts # Hardened Docker runner for generated probes
88
+ │ ├── index/ # Repo index, diff context, cache invalidation
89
+ │ ├── journal/ # Per-run structured journal entries
90
+ │ ├── probe/
91
+ │ │ ├── author.ts # Model-authored regression probe generation + validation
92
+ │ │ ├── harness.ts # vitest/jest/node:test detection + TAP classification
93
+ │ │ └── queue.ts # Head-vs-merge-base probe orchestration
70
94
  │ ├── report/
71
95
  │ │ ├── comment.ts # Markdown PR comment + commit-status rendering
72
96
  │ │ ├── junit.ts # JUnit XML output
package/action/action.yml CHANGED
@@ -46,6 +46,14 @@ inputs:
46
46
  description: Enable the B.2 probe lane — authored test probes for unexercised findings run in a hardened Docker container (no network, no secrets, read-only fs). Enable-only — 'false' does not override a config-enabled lane. Requires Docker on the runner; forced off on pull_request_target.
47
47
  default: 'false'
48
48
  required: false
49
+ max-comments:
50
+ description: Cap on inline review comments posted per run; overflow is summarized in the sticky. Overrides review.maxComments when set.
51
+ default: ''
52
+ required: false
53
+ run:
54
+ description: Run the browser-flow lane (`argus-reviewer run`). Set 'false' for code-review-only consumers — repos without a per-PR web target (CLIs, libraries, infra repos). Skips the Playwright install and run steps; the sticky comment and commit status then reflect code-review alone.
55
+ default: 'true'
56
+ required: false
49
57
  outputs:
50
58
  conclusion:
51
59
  description: 'Run conclusion: success, failure, or neutral'
@@ -65,6 +73,7 @@ runs:
65
73
  run: npm ci
66
74
 
67
75
  - name: Install Playwright browser
76
+ if: inputs.run != 'false'
68
77
  shell: bash
69
78
  working-directory: ${{ inputs.working-directory }}
70
79
  env:
@@ -93,6 +102,10 @@ runs:
93
102
  if: inputs.index == 'true'
94
103
  shell: bash
95
104
  working-directory: ${{ inputs.working-directory }}
105
+ env:
106
+ # Token only feeds trust resolution on issue_comment events —
107
+ # pull_request* events read fork status from the event payload.
108
+ GITHUB_TOKEN: ${{ github.token }}
96
109
  run: ${{ inputs.cli }} index
97
110
 
98
111
  - name: Run argus-reviewer code review
@@ -104,6 +117,7 @@ runs:
104
117
  OPENROUTER_API_KEY: ${{ inputs.openrouter-api-key }}
105
118
  GITHUB_TOKEN: ${{ github.token }}
106
119
  ARGUS_SANDBOX: ${{ inputs.sandbox == 'true' && '1' || '' }}
120
+ ARGUS_MAX_COMMENTS: ${{ inputs.max-comments }}
107
121
  ARGUS_DEBUG: '1'
108
122
  ARGUS_REVIEWER_TRACE: >-
109
123
  {"repo":"${{ github.repository }}",
@@ -116,11 +130,13 @@ runs:
116
130
 
117
131
  - name: Run argus-reviewer
118
132
  id: run
133
+ if: inputs.run != 'false'
119
134
  shell: bash
120
135
  working-directory: ${{ inputs.working-directory }}
121
136
  continue-on-error: true
122
137
  env:
123
138
  OPENROUTER_API_KEY: ${{ inputs.openrouter-api-key }}
139
+ GITHUB_TOKEN: ${{ github.token }}
124
140
  ARGUS_DEBUG: '1'
125
141
  ARGUS_DIFF_BASE: ${{ inputs.diff-base }}
126
142
  ARGUS_BUDGET_USD: ${{ inputs.budget-usd }}
@@ -140,6 +156,7 @@ runs:
140
156
  OPENROUTER_API_KEY: ${{ inputs.openrouter-api-key }}
141
157
  VISION_E2E_WORKING_DIR: ${{ inputs.working-directory }}
142
158
  ARGUS_REPORT_DIR: ${{ inputs.report-dir }}
159
+ ARGUS_RUN_DISABLED: ${{ inputs.run == 'false' && '1' || '' }}
143
160
  with:
144
161
  github-token: ${{ github.token }}
145
162
  script: |
@@ -11,7 +11,10 @@ function formatUsd(n) {
11
11
 
12
12
  /** Escape a report string for one markdown table cell. */
13
13
  function cell(s) {
14
- return String(s ?? '').replace(/\|/g, '\\|').replace(/[\r\n]+/g, ' ').slice(0, 200)
14
+ return String(s ?? '')
15
+ .replace(/\|/g, '\\|')
16
+ .replace(/[\r\n]+/g, ' ')
17
+ .slice(0, 200)
15
18
  }
16
19
 
17
20
  function renderMissingKeyBody() {
@@ -20,7 +23,9 @@ function renderMissingKeyBody() {
20
23
  lines.push('')
21
24
  lines.push('## argus-reviewer ⚪ skipped')
22
25
  lines.push('')
23
- lines.push('`OPENROUTER_API_KEY` is not configured. Add it as a repository or workflow secret to run argus-reviewer.')
26
+ lines.push(
27
+ '`OPENROUTER_API_KEY` is not configured. Add it as a repository or workflow secret to run argus-reviewer.',
28
+ )
24
29
  lines.push('')
25
30
  lines.push('This status is intentionally neutral, not a failure.')
26
31
  lines.push('')
@@ -33,14 +38,16 @@ function renderNoReportBody(reportDir, runUrl) {
33
38
  lines.push('')
34
39
  lines.push('## argus-reviewer ⚠️ no report')
35
40
  lines.push('')
36
- lines.push(`The run step produced no \`run.json\` under \`${reportDir}\`. The commit status fails closed — check the action logs before merging.`)
41
+ lines.push(
42
+ `The run step produced no \`run.json\` under \`${reportDir}\`. The commit status fails closed — check the action logs before merging.`,
43
+ )
37
44
  lines.push('')
38
45
  lines.push(`[View run](${runUrl})`)
39
46
  lines.push('')
40
47
  return lines.join('\n')
41
48
  }
42
49
 
43
- function renderBody(report, codeReview, runUrl, ok) {
50
+ function renderBody(report, codeReview, runUrl, ok, inlinePlan) {
44
51
  if (!report) return renderMissingKeyBody()
45
52
 
46
53
  const lines = []
@@ -73,7 +80,9 @@ function renderBody(report, codeReview, runUrl, ok) {
73
80
  lines.push(`- \`${path.basename(t.file)}\` — ${t.name}`)
74
81
  }
75
82
  lines.push('')
76
- lines.push(`**Risk:** ${ok ? 'Low — UI regression tests and code review passed; no heals or failures.' : 'High — investigate failures before merge.'}`)
83
+ lines.push(
84
+ `**Risk:** ${ok ? 'Low — UI regression tests and code review passed; no heals or failures.' : 'High — investigate failures before merge.'}`,
85
+ )
77
86
  lines.push('')
78
87
  if (Object.keys(trace).length > 0) {
79
88
  lines.push('**Trace**')
@@ -92,7 +101,9 @@ function renderBody(report, codeReview, runUrl, ok) {
92
101
  lines.push('| --- | --- | ---: | ---: | ---: | ---: |')
93
102
  for (const t of report.tests) {
94
103
  const result = t.ok ? '✅ pass' : '❌ fail'
95
- lines.push(`| ${t.name} | ${result} | ${t.visionCalls} | ${formatUsd(t.visionCostUsd)} | ${t.healEvents?.length ?? 0} | ${t.asserts?.length ?? 0} |`)
104
+ lines.push(
105
+ `| ${t.name} | ${result} | ${t.visionCalls} | ${formatUsd(t.visionCostUsd)} | ${t.healEvents?.length ?? 0} | ${t.asserts?.length ?? 0} |`,
106
+ )
96
107
  }
97
108
  lines.push('')
98
109
  lines.push('</details>')
@@ -178,14 +189,24 @@ function renderBody(report, codeReview, runUrl, ok) {
178
189
  lines.push('')
179
190
  lines.push('| Check | Status | Explanation |')
180
191
  lines.push('| --- | --- | --- |')
181
- lines.push(`| Tests | ${report.ok ? '✅ Passed' : '❌ Failed'} | ${report.totals.passed}/${report.totals.tests} tests passed |`)
182
- lines.push(`| Budget | ${report.totals.budgetExceeded ? '⚠️ Warning' : '✅ Passed'} | ${formatUsd(report.totals.visionCostUsd)} spent${budgetCap > 0 ? ` of ${formatUsd(budgetCap)}` : ''} |`)
183
- lines.push(`| Heal events | ${healCount === 0 ? '✅ Passed' : '⚠️ Warning'} | ${healCount} heal event${healCount === 1 ? '' : 's'} |`)
184
- lines.push(`| Assertions | ${assertFails === 0 ? '✅ Passed' : '❌ Failed'} | ${assertFails === 0 ? assertCount : `${assertFails} failed`} assertion${assertCount === 1 ? '' : 's'} |`)
192
+ lines.push(
193
+ `| Tests | ${report.ok ? '✅ Passed' : '❌ Failed'} | ${report.totals.passed}/${report.totals.tests} tests passed |`,
194
+ )
195
+ lines.push(
196
+ `| Budget | ${report.totals.budgetExceeded ? '⚠️ Warning' : '✅ Passed'} | ${formatUsd(report.totals.visionCostUsd)} spent${budgetCap > 0 ? ` of ${formatUsd(budgetCap)}` : ''} |`,
197
+ )
198
+ lines.push(
199
+ `| Heal events | ${healCount === 0 ? '✅ Passed' : '⚠️ Warning'} | ${healCount} heal event${healCount === 1 ? '' : 's'} |`,
200
+ )
201
+ lines.push(
202
+ `| Assertions | ${assertFails === 0 ? '✅ Passed' : '❌ Failed'} | ${assertFails === 0 ? assertCount : `${assertFails} failed`} assertion${assertCount === 1 ? '' : 's'} |`,
203
+ )
185
204
  lines.push(`| OpenRouter key | ✅ Passed | \`OPENROUTER_API_KEY\` configured |`)
186
205
  if (codeReview && !codeReview.skipped) {
187
206
  const codeStatus = codeReview.ok ? '✅ Passed' : '❌ Failed'
188
- lines.push(`| Code review | ${codeStatus} | ${codeReview.findings.length} findings (${codeReview.model}) |`)
207
+ lines.push(
208
+ `| Code review | ${codeStatus} | ${codeReview.findings.length} findings (${codeReview.model}) |`,
209
+ )
189
210
  } else {
190
211
  lines.push(`| Code review | ⚪ Skipped | ${codeReview?.summary ?? 'no report'} |`)
191
212
  }
@@ -193,43 +214,7 @@ function renderBody(report, codeReview, runUrl, ok) {
193
214
  lines.push('</details>')
194
215
  lines.push('')
195
216
 
196
- if (codeReview && !codeReview.skipped) {
197
- lines.push('<details>')
198
- lines.push('<summary>🧠 Code review</summary>')
199
- lines.push('')
200
- lines.push(`**Verdict:** ${codeReview.verdict} · ${codeReview.model} · ${codeReview.tokens}tok ${formatUsd(codeReview.visionCostUsd)}`)
201
- if (Array.isArray(codeReview.probes) && codeReview.probes.length > 0) {
202
- const reproduced = codeReview.probes.filter((p) => p.outcome === 'reproduced').length
203
- lines.push(` · 🧪 ${codeReview.probes.length} probe${codeReview.probes.length === 1 ? '' : 's'} run, ${reproduced} reproduced`)
204
- }
205
- if (typeof codeReview.probeLaneSkipped === 'string') {
206
- lines.push(` · 🧪 probe lane skipped — ${cell(codeReview.probeLaneSkipped)}`)
207
- }
208
- lines.push('')
209
- lines.push(codeReview.summary)
210
- lines.push('')
211
- if (codeReview.findings.length > 0) {
212
- const evidenceIcon = { exercised: '✅', corroborated: '🔴', not_exercised: '⚪', inconclusive: '❔', reproduced: '🧪' }
213
- lines.push('| File | Severity | Evidence | Finding |')
214
- lines.push('| --- | --- | --- | --- |')
215
- // Findings/evidence strings are model- and probe-emitted — sanitize
216
- // for the markdown table and bound the section so an oversized report
217
- // can't push the body past GitHub's 65536-char comment limit.
218
- const MAX_FINDING_ROWS = 25
219
- for (const f of codeReview.findings.slice(0, MAX_FINDING_ROWS)) {
220
- const ev = f.evidence
221
- ? `${evidenceIcon[f.evidence.status] ?? '❔'} ${cell(f.evidence.detail)}`
222
- : '—'
223
- lines.push(`| \`${cell(f.file)}\` | ${cell(f.severity)} | ${ev} | ${cell(f.message)} |`)
224
- }
225
- if (codeReview.findings.length > MAX_FINDING_ROWS) {
226
- lines.push(`| … | — | — | ${codeReview.findings.length - MAX_FINDING_ROWS} more findings in \`code-review.json\` |`)
227
- }
228
- lines.push('')
229
- }
230
- lines.push('</details>')
231
- lines.push('')
232
- }
217
+ pushCodeReviewDetails(lines, codeReview, inlinePlan)
233
218
 
234
219
  lines.push('<details>')
235
220
  lines.push('<summary>✨ Actions</summary>')
@@ -247,50 +232,182 @@ function renderBody(report, codeReview, runUrl, ok) {
247
232
  return lines.join('\n')
248
233
  }
249
234
 
250
- async function main() {
251
- const pr = context.payload && context.payload.pull_request
252
- const owner = context.repo.owner
253
- const repo = context.repo.repo
254
- const hasKey = !!process.env.OPENROUTER_API_KEY
255
- const workDir = process.env.VISION_E2E_WORKING_DIR || ''
256
- const reportDir = path.resolve(
257
- process.env.GITHUB_WORKSPACE,
258
- workDir,
259
- process.env.ARGUS_REPORT_DIR || 'argus-reviewer-report',
235
+ // The 🧠 Code review details block — shared by the full body and the
236
+ // review-only body (run lane disabled).
237
+ function pushCodeReviewDetails(lines, codeReview, inlinePlan) {
238
+ if (!codeReview || codeReview.skipped) return
239
+ lines.push('<details>')
240
+ lines.push('<summary>🧠 Code review</summary>')
241
+ lines.push('')
242
+ lines.push(
243
+ `**Verdict:** ${codeReview.verdict} · ${codeReview.model} · ${codeReview.tokens}tok ${formatUsd(codeReview.visionCostUsd)}`,
260
244
  )
261
- const runUrl = `${process.env.GITHUB_SERVER_URL}/${owner}/${repo}/actions/runs/${process.env.GITHUB_RUN_ID}`
262
-
263
- let report
264
- let codeReview
265
- if (hasKey) {
266
- try {
267
- const raw = fs.readFileSync(path.join(reportDir, 'run.json'), 'utf8')
268
- report = JSON.parse(raw)
269
- } catch {
270
- report = undefined
245
+ // U7 triage record — Jev annotate/route signals, never the gate.
246
+ if (codeReview.triage) {
247
+ const t = codeReview.triage
248
+ if (t.unadjudicated === true) {
249
+ lines.push(` · 🧭 triage unadjudicated — Jev unavailable`)
250
+ } else {
251
+ lines.push(
252
+ ` · 🧭 triage: risk ${t.risk ?? '?'}/5` +
253
+ `${typeof t.needsDeepReview === 'number' ? ` · deep-review ${t.needsDeepReview.toFixed(2)}` : ''}` +
254
+ `${t.topRiskArea !== undefined ? ` · top area \`${cell(t.topRiskArea)}\`` : ''}` +
255
+ ` (${cell(t.mode)})`,
256
+ )
271
257
  }
272
- try {
273
- const raw = fs.readFileSync(path.join(reportDir, 'code-review.json'), 'utf8')
274
- codeReview = JSON.parse(raw)
275
- } catch {
276
- codeReview = undefined
258
+ }
259
+ if (Array.isArray(codeReview.probes) && codeReview.probes.length > 0) {
260
+ const reproduced = codeReview.probes.filter((p) => p.outcome === 'reproduced').length
261
+ lines.push(
262
+ ` · 🧪 ${codeReview.probes.length} probe${codeReview.probes.length === 1 ? '' : 's'} run, ${reproduced} reproduced`,
263
+ )
264
+ }
265
+ if (typeof codeReview.probeLaneSkipped === 'string') {
266
+ lines.push(` · 🧪 probe lane skipped — ${cell(codeReview.probeLaneSkipped)}`)
267
+ }
268
+ lines.push('')
269
+ lines.push(codeReview.summary)
270
+ lines.push('')
271
+ if (codeReview.findings.length > 0) {
272
+ const evidenceIcon = {
273
+ exercised: '✅',
274
+ corroborated: '🔴',
275
+ not_exercised: '⚪',
276
+ inconclusive: '❔',
277
+ reproduced: '🧪',
278
+ }
279
+ lines.push('| File | Severity | p | Category | Evidence | Finding |')
280
+ lines.push('| --- | --- | --- | --- | --- | --- |')
281
+ // Findings/evidence strings are model- and probe-emitted — sanitize
282
+ // for the markdown table and bound the section so an oversized report
283
+ // can't push the body past GitHub's 65536-char comment limit.
284
+ const MAX_FINDING_ROWS = 25
285
+ for (const f of codeReview.findings.slice(0, MAX_FINDING_ROWS)) {
286
+ const ev = f.evidence
287
+ ? `${evidenceIcon[f.evidence.status] ?? '❔'} ${cell(f.evidence.detail)}`
288
+ : '—'
289
+ // U8 — Jev P(true positive); unadjudicated findings render '—'.
290
+ const p = typeof f.p === 'number' ? f.p.toFixed(2) : '—'
291
+ lines.push(
292
+ `| \`${cell(f.file)}\` | ${cell(f.severity)} | ${p} | ${cell(f.category ?? '—')} | ${ev} | ${cell(f.message)} |`,
293
+ )
294
+ }
295
+ if (codeReview.findings.length > MAX_FINDING_ROWS) {
296
+ lines.push(
297
+ `| … | — | — | — | — | ${codeReview.findings.length - MAX_FINDING_ROWS} more findings in \`code-review.json\` |`,
298
+ )
299
+ }
300
+ lines.push('')
301
+ // Inline-comment cap note — dedup'd fresh findings the
302
+ // review.maxComments budget didn't post (TCA max_comments).
303
+ if (inlinePlan !== undefined && inlinePlan.dropped > 0) {
304
+ lines.push(
305
+ `*+${inlinePlan.dropped} inline-eligible finding(s) not posted — \`review.maxComments\` cap ${inlinePlan.cap}.*`,
306
+ )
307
+ lines.push('')
308
+ }
309
+ // Secrets-lane audit line — adjudicated/suppressed counts, never literals.
310
+ if (codeReview.secretsScan) {
311
+ if (typeof codeReview.secretsScan.skipped === 'string') {
312
+ lines.push(`*🔐 secrets scan skipped — ${cell(codeReview.secretsScan.skipped)}*`)
313
+ } else if (Array.isArray(codeReview.secretsScan.records)) {
314
+ const suppressed = codeReview.secretsScan.records.filter((r) => r.suppressed).length
315
+ const unadj = codeReview.secretsScan.records.filter((r) => !r.adjudicated).length
316
+ lines.push(
317
+ `*🔐 secrets scan: ${codeReview.secretsScan.records.length} candidate(s)` +
318
+ `${suppressed > 0 ? `, ${suppressed} adjudicated-suppressed` : ''}` +
319
+ `${unadj > 0 ? `, ${unadj} unadjudicated` : ''}` +
320
+ `${codeReview.secretsScan.overflow > 0 ? `, +${codeReview.secretsScan.overflow} over cap` : ''}.*`,
321
+ )
322
+ }
323
+ lines.push('')
277
324
  }
278
325
  }
326
+ // U8 adjudication audit — outside the findings guard so suppressed-
327
+ // only reviews still show what Jev removed. p values live on the
328
+ // findings table and in code-review.json records.
329
+ if (codeReview.findingAdjudication && Array.isArray(codeReview.findingAdjudication.records)) {
330
+ const fa = codeReview.findingAdjudication
331
+ if (fa.unadjudicated === true) {
332
+ lines.push('*🧮 adjudication: unadjudicated — Jev unavailable, nothing suppressed.*')
333
+ } else {
334
+ const suppressed = fa.records.filter((r) => r.suppressed).length
335
+ const unadj = fa.records.filter((r) => !r.adjudicated).length
336
+ lines.push(
337
+ `*🧮 adjudication: ${fa.records.length} finding(s) scored` +
338
+ `${suppressed > 0 ? `, ${suppressed} suppressed (nit/q)` : ''}` +
339
+ `${unadj > 0 ? `, ${unadj} unadjudicated` : ''}` +
340
+ `${fa.overflow > 0 ? `, +${fa.overflow} over cap` : ''}.*`,
341
+ )
342
+ }
343
+ lines.push('')
344
+ }
345
+ lines.push('</details>')
346
+ lines.push('')
347
+ }
279
348
 
280
- // Missing code-review.json after a continue-on-error step means the review
281
- // crashed, not that it skipped — an intentional skip writes ok+skipped.
282
- // Fail closed rather than reporting it as a clean skip.
283
- const codeReviewOk = codeReview != null && codeReview.ok === true
284
- const ok = (report?.ok === true) && codeReviewOk
285
- const conclusion = !hasKey ? 'neutral' : ok ? 'success' : 'failure'
286
- const body = !hasKey
287
- ? renderMissingKeyBody()
288
- : report === undefined
289
- ? renderNoReportBody(reportDir, runUrl)
290
- : renderBody(report, codeReview, runUrl, ok)
349
+ // Sticky body for `run: 'false'` consumers — no run.json exists by
350
+ // design, so the body and conclusion reflect code-review alone.
351
+ function renderReviewOnlyBody(codeReview, runUrl, ok, inlinePlan) {
352
+ const lines = []
353
+ lines.push(SENTINEL)
354
+ lines.push('')
355
+ lines.push(`## argus-reviewer ${ok ? '✅ PASS' : '❌ FAIL'}`)
356
+ lines.push('')
357
+ if (!codeReview) {
358
+ lines.push(
359
+ '**Summary:** code-review only (run lane disabled) — no `code-review.json` found. The review step crashed or produced no report; the commit status fails closed — check the action logs before merging.',
360
+ )
361
+ } else if (codeReview.skipped) {
362
+ lines.push(
363
+ `**Summary:** code-review only (run lane disabled) — review skipped: ${cell(codeReview.summary)}`,
364
+ )
365
+ } else {
366
+ lines.push(
367
+ `**Summary:** code review only (run lane disabled) · verdict **${codeReview.verdict}** · ` +
368
+ `${codeReview.findings.length} finding(s) · ${codeReview.model} · ` +
369
+ `${codeReview.tokens}tok ${formatUsd(codeReview.visionCostUsd)}`,
370
+ )
371
+ }
372
+ lines.push('')
373
+ lines.push('<details>')
374
+ lines.push('<summary>🚥 Pre-merge checks</summary>')
375
+ lines.push('')
376
+ lines.push('| Check | Status | Explanation |')
377
+ lines.push('| --- | --- | --- |')
378
+ lines.push('| OpenRouter key | ✅ Passed | `OPENROUTER_API_KEY` configured |')
379
+ if (codeReview && !codeReview.skipped) {
380
+ const codeStatus = codeReview.ok ? '✅ Passed' : '❌ Failed'
381
+ lines.push(
382
+ `| Code review | ${codeStatus} | ${codeReview.findings.length} findings (${codeReview.model}) |`,
383
+ )
384
+ } else {
385
+ lines.push(
386
+ `| Code review | ${codeReview ? '⚪ Skipped' : '❌ Failed'} | ${cell(codeReview?.summary ?? 'no report')} |`,
387
+ )
388
+ }
389
+ lines.push('')
390
+ lines.push('</details>')
391
+ lines.push('')
392
+ pushCodeReviewDetails(lines, codeReview, inlinePlan)
393
+ if (runUrl) lines.push(`[View run](${runUrl})`)
394
+ lines.push('')
395
+ lines.push('<details>')
396
+ lines.push('<summary>✨ Actions</summary>')
397
+ lines.push('')
398
+ lines.push('- [ ] Re-run argus-reviewer')
399
+ lines.push('')
400
+ lines.push('</details>')
401
+ lines.push('')
402
+ lines.push('---')
403
+ lines.push('')
404
+ lines.push('<sub>`argus-reviewer` — self-hosted, BYOK OpenRouter UI regression.</sub>')
405
+ lines.push('')
406
+ return lines.join('\n')
407
+ }
291
408
 
292
- async function postInlineComments(pr, codeReview) {
293
- if (!pr || !codeReview || codeReview.skipped || !codeReview.findings) return
409
+ async function planInlineComments(pr, codeReview) {
410
+ if (!pr || !codeReview || codeReview.skipped || !codeReview.findings) return undefined
294
411
  // Must match the severity vocabulary emitted by the code-review schema
295
412
  // (src/cli.ts): bug/risk are inline-worthy; nit/q stay in the sticky body.
296
413
  const inlineSeverities = ['bug', 'risk']
@@ -301,6 +418,8 @@ async function postInlineComments(pr, codeReview) {
301
418
  line: f.line,
302
419
  side: 'RIGHT',
303
420
  body: `**argus-reviewer ${f.severity}:** ${f.message}${
421
+ f.category ? ` \`${f.category}\`` : ''
422
+ }${
304
423
  f.evidence && f.evidence.status === 'reproduced'
305
424
  ? '\n\n*🧪 Reproduced by an Argus probe — fails on this PR head, clean on base. See workflow artifacts.*'
306
425
  : f.evidence && f.evidence.status !== 'exercised'
@@ -308,7 +427,7 @@ async function postInlineComments(pr, codeReview) {
308
427
  : ''
309
428
  }`,
310
429
  }))
311
- if (comments.length === 0) return
430
+ if (comments.length === 0) return { capped: [], dropped: 0, cap: 0 }
312
431
 
313
432
  // Re-runs on the same SHA must not duplicate inline comments — the sticky
314
433
  // body is upserted but review comments are not. Paginate fully (100/page)
@@ -336,8 +455,17 @@ async function postInlineComments(pr, codeReview) {
336
455
  page += 1
337
456
  }
338
457
  const fresh = comments.filter((c) => !posted.has(dedupKey(c.path, c.line, c.body)))
339
- if (fresh.length === 0) return
340
458
 
459
+ // review.maxComments caps inline noise (TCA max_comments) — applied
460
+ // after dedup so already-posted comments don't eat the budget. A cap of
461
+ // 0 disables inline posting entirely (no empty review).
462
+ const cap = typeof codeReview.maxComments === 'number' ? codeReview.maxComments : 20
463
+ const capped = fresh.slice(0, cap)
464
+ return { capped, dropped: fresh.length - capped.length, cap }
465
+ }
466
+
467
+ async function postInlineComments(pr, plan) {
468
+ if (plan === undefined || plan.capped.length === 0) return
341
469
  // One batched review instead of N createReviewComment calls — avoids
342
470
  // secondary rate limits on large findings sets.
343
471
  try {
@@ -347,13 +475,65 @@ async function postInlineComments(pr, codeReview) {
347
475
  pull_number: pr.number,
348
476
  commit_id: pr.head.sha,
349
477
  event: 'COMMENT',
350
- comments: fresh,
478
+ comments: plan.capped,
351
479
  })
352
480
  } catch (e) {
353
481
  core.warning(`inline review failed: ${e.message}`)
354
482
  }
355
483
  }
356
484
 
485
+ async function main() {
486
+ const pr = context.payload && context.payload.pull_request
487
+ const owner = context.repo.owner
488
+ const repo = context.repo.repo
489
+ const hasKey = !!process.env.OPENROUTER_API_KEY
490
+ const workDir = process.env.VISION_E2E_WORKING_DIR || ''
491
+ const reportDir = path.resolve(
492
+ process.env.GITHUB_WORKSPACE,
493
+ workDir,
494
+ process.env.ARGUS_REPORT_DIR || 'argus-reviewer-report',
495
+ )
496
+ const runUrl = `${process.env.GITHUB_SERVER_URL}/${owner}/${repo}/actions/runs/${process.env.GITHUB_RUN_ID}`
497
+
498
+ let report
499
+ let codeReview
500
+ if (hasKey) {
501
+ try {
502
+ const raw = fs.readFileSync(path.join(reportDir, 'run.json'), 'utf8')
503
+ report = JSON.parse(raw)
504
+ } catch {
505
+ report = undefined
506
+ }
507
+ try {
508
+ const raw = fs.readFileSync(path.join(reportDir, 'code-review.json'), 'utf8')
509
+ codeReview = JSON.parse(raw)
510
+ } catch {
511
+ codeReview = undefined
512
+ }
513
+ }
514
+
515
+ // Missing code-review.json after a continue-on-error step means the review
516
+ // crashed, not that it skipped — an intentional skip writes ok+skipped.
517
+ // Fail closed rather than reporting it as a clean skip.
518
+ const codeReviewOk = codeReview != null && codeReview.ok === true
519
+ // run: 'false' consumers have no run.json by design — the conclusion then
520
+ // reflects the code-review verdict alone.
521
+ const runDisabled = process.env.ARGUS_RUN_DISABLED === '1'
522
+ const ok = (runDisabled || report?.ok === true) && codeReviewOk
523
+ const conclusion = !hasKey ? 'neutral' : ok ? 'success' : 'failure'
524
+ const inlinePlan = hasKey ? await planInlineComments(pr, codeReview) : undefined
525
+ const body = !hasKey
526
+ ? renderMissingKeyBody()
527
+ : runDisabled
528
+ ? renderReviewOnlyBody(codeReview, runUrl, ok, inlinePlan)
529
+ : report === undefined
530
+ ? renderNoReportBody(reportDir, runUrl)
531
+ : renderBody(report, codeReview, runUrl, ok, inlinePlan)
532
+
533
+ // Eligibility + dedup for inline comments, computed before the sticky
534
+ // body renders so the "+N not posted" note counts the *fresh* set — the
535
+ // cap is applied to fresh, not to raw findings (already-posted comments
536
+ // must not inflate the dropped count).
357
537
  if (pr) {
358
538
  const { data: comments } = await github.rest.issues.listComments({
359
539
  owner,
@@ -377,7 +557,7 @@ async function postInlineComments(pr, codeReview) {
377
557
  body,
378
558
  })
379
559
  }
380
- await postInlineComments(pr, codeReview)
560
+ await postInlineComments(pr, inlinePlan)
381
561
  }
382
562
 
383
563
  const sha = pr ? pr.head.sha : context.sha
package/dist/api.d.ts CHANGED
@@ -7,6 +7,7 @@ import { Config, defineConfig } from './config.js';
7
7
  import { ErrorRecord } from './journal/schema.js';
8
8
  import { Logger } from './log.js';
9
9
  export { defineConfig };
10
+ export { DecisionClient, DecisionError, JEV_DEFAULT_MODEL, type DecisionAnswer, type DecisionClientOptions, type DecisionErrorKind, type DecisionQuestion, } from './vision/decisions.js';
10
11
  /**
11
12
  * Test-facing API (R13). Test files are plain TypeScript using a `td` object:
12
13
  *
package/dist/api.js CHANGED
@@ -4,6 +4,7 @@ import { loadFlow, saveFlow } from './cache/store.js';
4
4
  import { Ledger } from './vision/ledger.js';
5
5
  import { defineConfig } from './config.js';
6
6
  export { defineConfig };
7
+ export { DecisionClient, DecisionError, JEV_DEFAULT_MODEL, } from './vision/decisions.js';
7
8
  const KEY_ALIASES = {
8
9
  tab: 'Tab',
9
10
  enter: 'Enter',