argus-reviewer-e2e 0.1.3 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +29 -5
- package/action/action.yml +17 -0
- package/action/sticky-comment.mjs +270 -90
- package/dist/api.d.ts +1 -0
- package/dist/api.js +1 -0
- package/dist/cli.d.ts +67 -0
- package/dist/cli.js +383 -49
- package/dist/config.d.ts +63 -2
- package/dist/config.js +108 -5
- package/dist/debug.d.ts +1 -0
- package/dist/debug.js +9 -3
- package/dist/evidence/ci.d.ts +3 -0
- package/dist/evidence/ci.js +3 -1
- package/dist/probe/queue.d.ts +4 -1
- package/dist/probe/queue.js +46 -8
- package/dist/review/adjudicate.d.ts +63 -0
- package/dist/review/adjudicate.js +111 -0
- package/dist/review/secrets.d.ts +88 -0
- package/dist/review/secrets.js +220 -0
- package/dist/review/triage.d.ts +76 -0
- package/dist/review/triage.js +163 -0
- package/dist/trust.d.ts +50 -0
- package/dist/trust.js +103 -0
- package/dist/vision/cost.d.ts +14 -1
- package/dist/vision/cost.js +13 -0
- package/dist/vision/decisions.d.ts +95 -0
- package/dist/vision/decisions.js +232 -0
- package/package.json +3 -2
package/README.md
CHANGED
|
@@ -4,6 +4,13 @@
|
|
|
4
4
|
<img src="docs/assets/social.png" alt="Argus — vision-model E2E testing" width="640" />
|
|
5
5
|
</p>
|
|
6
6
|
|
|
7
|
+
<p align="center">
|
|
8
|
+
<a href="https://www.npmjs.com/package/argus-reviewer-e2e"><img src="https://img.shields.io/npm/v/argus-reviewer-e2e" alt="npm version" /></a>
|
|
9
|
+
<a href="https://github.com/duketopceo/Argus/actions/workflows/ci.yml"><img src="https://github.com/duketopceo/Argus/actions/workflows/ci.yml/badge.svg" alt="CI" /></a>
|
|
10
|
+
<a href="LICENSE"><img src="https://img.shields.io/badge/license-MIT-blue" alt="MIT license" /></a>
|
|
11
|
+
<a href="https://github.com/duketopceo/Argus/security/policy"><img src="https://img.shields.io/badge/security-policy-orange" alt="security policy" /></a>
|
|
12
|
+
</p>
|
|
13
|
+
|
|
7
14
|
Open-source, self-hosted vision-model E2E testing — the hundred-eyed watcher for your UI. Bring your own `OPENROUTER_API_KEY`: record a flow once, fingerprint-cache every step, replay near-free, heal on UI drift, and get results as a check + comment on the GitHub PR.
|
|
8
15
|
|
|
9
16
|
- **Vision-first**: a model looks at a screenshot and decides where to click — no selectors to write or maintain.
|
|
@@ -41,7 +48,10 @@ export default {
|
|
|
41
48
|
|
|
42
49
|
The GitHub Action automatically sets `ARGUS_REVIEWER_TRACE` with the repository, PR number, commit, and run id, so every PR review is attributed in OpenRouter without extra config. You can also set `ARGUS_REVIEWER_TRACE` yourself (JSON object) to add more fields.
|
|
43
50
|
|
|
44
|
-
Status: early development. See `action/` for the composite GitHub Action
|
|
51
|
+
Status: early development. See `action/` for the composite GitHub Action,
|
|
52
|
+
`runner/` for self-hosted runner registration, `docs/quickstart.md` for
|
|
53
|
+
setup, `SECURITY.md` for the threat model, and `CONTRIBUTING.md` to hack
|
|
54
|
+
on it.
|
|
45
55
|
|
|
46
56
|
## File structure
|
|
47
57
|
|
|
@@ -53,10 +63,14 @@ argus-reviewer/
|
|
|
53
63
|
├── runner/ # Self-hosted runner registration docs + script
|
|
54
64
|
│ ├── README.md
|
|
55
65
|
│ └── register-runner.sh
|
|
66
|
+
├── electron/ # Local observability dashboard (`npm run app`)
|
|
56
67
|
├── src/
|
|
57
68
|
│ ├── api.ts # Test-facing `test`/`td` API + generated test file renderer
|
|
58
|
-
│ ├── cli.ts #
|
|
69
|
+
│ ├── cli.ts # record · run · code-review · delegate · cache · index · init
|
|
59
70
|
│ ├── config.ts # `argus-reviewer.config.*` loader (legacy `vision-e2e.config.*` accepted)
|
|
71
|
+
│ ├── cache/
|
|
72
|
+
│ │ ├── fingerprint.ts # Per-step screenshot/a11y fingerprint + resolve
|
|
73
|
+
│ │ └── store.ts # Flow cache read/write
|
|
60
74
|
│ ├── driver/
|
|
61
75
|
│ │ ├── browser.ts # Playwright browser launch (chromium/firefox/webkit) + observation capture
|
|
62
76
|
│ │ └── target.ts # Optional local dev-server target process
|
|
@@ -64,9 +78,19 @@ argus-reviewer/
|
|
|
64
78
|
│ │ ├── actions.ts # Low-level page actions (click, type, scroll, …)
|
|
65
79
|
│ │ ├── loop.ts # Vision model record/replay + healing loop
|
|
66
80
|
│ │ └── prompts.ts # OpenRouter action/assertion prompts + JSON schemas
|
|
67
|
-
│ ├──
|
|
68
|
-
│ │ ├──
|
|
69
|
-
│ │
|
|
81
|
+
│ ├── evidence/
|
|
82
|
+
│ │ ├── ci.ts # PR metadata + CI check-run context for findings
|
|
83
|
+
│ │ ├── gate.ts # Fork-PR trust gate (argus-probe label bound to head SHA)
|
|
84
|
+
│ │ └── link.ts # Finding → evidence linkage + comment-safe sanitization
|
|
85
|
+
│ ├── executor/
|
|
86
|
+
│ │ ├── a0.ts # `a0 headless -p` delegation to a user's Agent Zero instance
|
|
87
|
+
│ │ └── sandbox.ts # Hardened Docker runner for generated probes
|
|
88
|
+
│ ├── index/ # Repo index, diff context, cache invalidation
|
|
89
|
+
│ ├── journal/ # Per-run structured journal entries
|
|
90
|
+
│ ├── probe/
|
|
91
|
+
│ │ ├── author.ts # Model-authored regression probe generation + validation
|
|
92
|
+
│ │ ├── harness.ts # vitest/jest/node:test detection + TAP classification
|
|
93
|
+
│ │ └── queue.ts # Head-vs-merge-base probe orchestration
|
|
70
94
|
│ ├── report/
|
|
71
95
|
│ │ ├── comment.ts # Markdown PR comment + commit-status rendering
|
|
72
96
|
│ │ ├── junit.ts # JUnit XML output
|
package/action/action.yml
CHANGED
|
@@ -46,6 +46,14 @@ inputs:
|
|
|
46
46
|
description: Enable the B.2 probe lane — authored test probes for unexercised findings run in a hardened Docker container (no network, no secrets, read-only fs). Enable-only — 'false' does not override a config-enabled lane. Requires Docker on the runner; forced off on pull_request_target.
|
|
47
47
|
default: 'false'
|
|
48
48
|
required: false
|
|
49
|
+
max-comments:
|
|
50
|
+
description: Cap on inline review comments posted per run; overflow is summarized in the sticky. Overrides review.maxComments when set.
|
|
51
|
+
default: ''
|
|
52
|
+
required: false
|
|
53
|
+
run:
|
|
54
|
+
description: Run the browser-flow lane (`argus-reviewer run`). Set 'false' for code-review-only consumers — repos without a per-PR web target (CLIs, libraries, infra repos). Skips the Playwright install and run steps; the sticky comment and commit status then reflect code-review alone.
|
|
55
|
+
default: 'true'
|
|
56
|
+
required: false
|
|
49
57
|
outputs:
|
|
50
58
|
conclusion:
|
|
51
59
|
description: 'Run conclusion: success, failure, or neutral'
|
|
@@ -65,6 +73,7 @@ runs:
|
|
|
65
73
|
run: npm ci
|
|
66
74
|
|
|
67
75
|
- name: Install Playwright browser
|
|
76
|
+
if: inputs.run != 'false'
|
|
68
77
|
shell: bash
|
|
69
78
|
working-directory: ${{ inputs.working-directory }}
|
|
70
79
|
env:
|
|
@@ -93,6 +102,10 @@ runs:
|
|
|
93
102
|
if: inputs.index == 'true'
|
|
94
103
|
shell: bash
|
|
95
104
|
working-directory: ${{ inputs.working-directory }}
|
|
105
|
+
env:
|
|
106
|
+
# Token only feeds trust resolution on issue_comment events —
|
|
107
|
+
# pull_request* events read fork status from the event payload.
|
|
108
|
+
GITHUB_TOKEN: ${{ github.token }}
|
|
96
109
|
run: ${{ inputs.cli }} index
|
|
97
110
|
|
|
98
111
|
- name: Run argus-reviewer code review
|
|
@@ -104,6 +117,7 @@ runs:
|
|
|
104
117
|
OPENROUTER_API_KEY: ${{ inputs.openrouter-api-key }}
|
|
105
118
|
GITHUB_TOKEN: ${{ github.token }}
|
|
106
119
|
ARGUS_SANDBOX: ${{ inputs.sandbox == 'true' && '1' || '' }}
|
|
120
|
+
ARGUS_MAX_COMMENTS: ${{ inputs.max-comments }}
|
|
107
121
|
ARGUS_DEBUG: '1'
|
|
108
122
|
ARGUS_REVIEWER_TRACE: >-
|
|
109
123
|
{"repo":"${{ github.repository }}",
|
|
@@ -116,11 +130,13 @@ runs:
|
|
|
116
130
|
|
|
117
131
|
- name: Run argus-reviewer
|
|
118
132
|
id: run
|
|
133
|
+
if: inputs.run != 'false'
|
|
119
134
|
shell: bash
|
|
120
135
|
working-directory: ${{ inputs.working-directory }}
|
|
121
136
|
continue-on-error: true
|
|
122
137
|
env:
|
|
123
138
|
OPENROUTER_API_KEY: ${{ inputs.openrouter-api-key }}
|
|
139
|
+
GITHUB_TOKEN: ${{ github.token }}
|
|
124
140
|
ARGUS_DEBUG: '1'
|
|
125
141
|
ARGUS_DIFF_BASE: ${{ inputs.diff-base }}
|
|
126
142
|
ARGUS_BUDGET_USD: ${{ inputs.budget-usd }}
|
|
@@ -140,6 +156,7 @@ runs:
|
|
|
140
156
|
OPENROUTER_API_KEY: ${{ inputs.openrouter-api-key }}
|
|
141
157
|
VISION_E2E_WORKING_DIR: ${{ inputs.working-directory }}
|
|
142
158
|
ARGUS_REPORT_DIR: ${{ inputs.report-dir }}
|
|
159
|
+
ARGUS_RUN_DISABLED: ${{ inputs.run == 'false' && '1' || '' }}
|
|
143
160
|
with:
|
|
144
161
|
github-token: ${{ github.token }}
|
|
145
162
|
script: |
|
|
@@ -11,7 +11,10 @@ function formatUsd(n) {
|
|
|
11
11
|
|
|
12
12
|
/** Escape a report string for one markdown table cell. */
|
|
13
13
|
function cell(s) {
|
|
14
|
-
return String(s ?? '')
|
|
14
|
+
return String(s ?? '')
|
|
15
|
+
.replace(/\|/g, '\\|')
|
|
16
|
+
.replace(/[\r\n]+/g, ' ')
|
|
17
|
+
.slice(0, 200)
|
|
15
18
|
}
|
|
16
19
|
|
|
17
20
|
function renderMissingKeyBody() {
|
|
@@ -20,7 +23,9 @@ function renderMissingKeyBody() {
|
|
|
20
23
|
lines.push('')
|
|
21
24
|
lines.push('## argus-reviewer ⚪ skipped')
|
|
22
25
|
lines.push('')
|
|
23
|
-
lines.push(
|
|
26
|
+
lines.push(
|
|
27
|
+
'`OPENROUTER_API_KEY` is not configured. Add it as a repository or workflow secret to run argus-reviewer.',
|
|
28
|
+
)
|
|
24
29
|
lines.push('')
|
|
25
30
|
lines.push('This status is intentionally neutral, not a failure.')
|
|
26
31
|
lines.push('')
|
|
@@ -33,14 +38,16 @@ function renderNoReportBody(reportDir, runUrl) {
|
|
|
33
38
|
lines.push('')
|
|
34
39
|
lines.push('## argus-reviewer ⚠️ no report')
|
|
35
40
|
lines.push('')
|
|
36
|
-
lines.push(
|
|
41
|
+
lines.push(
|
|
42
|
+
`The run step produced no \`run.json\` under \`${reportDir}\`. The commit status fails closed — check the action logs before merging.`,
|
|
43
|
+
)
|
|
37
44
|
lines.push('')
|
|
38
45
|
lines.push(`[View run](${runUrl})`)
|
|
39
46
|
lines.push('')
|
|
40
47
|
return lines.join('\n')
|
|
41
48
|
}
|
|
42
49
|
|
|
43
|
-
function renderBody(report, codeReview, runUrl, ok) {
|
|
50
|
+
function renderBody(report, codeReview, runUrl, ok, inlinePlan) {
|
|
44
51
|
if (!report) return renderMissingKeyBody()
|
|
45
52
|
|
|
46
53
|
const lines = []
|
|
@@ -73,7 +80,9 @@ function renderBody(report, codeReview, runUrl, ok) {
|
|
|
73
80
|
lines.push(`- \`${path.basename(t.file)}\` — ${t.name}`)
|
|
74
81
|
}
|
|
75
82
|
lines.push('')
|
|
76
|
-
lines.push(
|
|
83
|
+
lines.push(
|
|
84
|
+
`**Risk:** ${ok ? 'Low — UI regression tests and code review passed; no heals or failures.' : 'High — investigate failures before merge.'}`,
|
|
85
|
+
)
|
|
77
86
|
lines.push('')
|
|
78
87
|
if (Object.keys(trace).length > 0) {
|
|
79
88
|
lines.push('**Trace**')
|
|
@@ -92,7 +101,9 @@ function renderBody(report, codeReview, runUrl, ok) {
|
|
|
92
101
|
lines.push('| --- | --- | ---: | ---: | ---: | ---: |')
|
|
93
102
|
for (const t of report.tests) {
|
|
94
103
|
const result = t.ok ? '✅ pass' : '❌ fail'
|
|
95
|
-
lines.push(
|
|
104
|
+
lines.push(
|
|
105
|
+
`| ${t.name} | ${result} | ${t.visionCalls} | ${formatUsd(t.visionCostUsd)} | ${t.healEvents?.length ?? 0} | ${t.asserts?.length ?? 0} |`,
|
|
106
|
+
)
|
|
96
107
|
}
|
|
97
108
|
lines.push('')
|
|
98
109
|
lines.push('</details>')
|
|
@@ -178,14 +189,24 @@ function renderBody(report, codeReview, runUrl, ok) {
|
|
|
178
189
|
lines.push('')
|
|
179
190
|
lines.push('| Check | Status | Explanation |')
|
|
180
191
|
lines.push('| --- | --- | --- |')
|
|
181
|
-
lines.push(
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
lines.push(
|
|
192
|
+
lines.push(
|
|
193
|
+
`| Tests | ${report.ok ? '✅ Passed' : '❌ Failed'} | ${report.totals.passed}/${report.totals.tests} tests passed |`,
|
|
194
|
+
)
|
|
195
|
+
lines.push(
|
|
196
|
+
`| Budget | ${report.totals.budgetExceeded ? '⚠️ Warning' : '✅ Passed'} | ${formatUsd(report.totals.visionCostUsd)} spent${budgetCap > 0 ? ` of ${formatUsd(budgetCap)}` : ''} |`,
|
|
197
|
+
)
|
|
198
|
+
lines.push(
|
|
199
|
+
`| Heal events | ${healCount === 0 ? '✅ Passed' : '⚠️ Warning'} | ${healCount} heal event${healCount === 1 ? '' : 's'} |`,
|
|
200
|
+
)
|
|
201
|
+
lines.push(
|
|
202
|
+
`| Assertions | ${assertFails === 0 ? '✅ Passed' : '❌ Failed'} | ${assertFails === 0 ? assertCount : `${assertFails} failed`} assertion${assertCount === 1 ? '' : 's'} |`,
|
|
203
|
+
)
|
|
185
204
|
lines.push(`| OpenRouter key | ✅ Passed | \`OPENROUTER_API_KEY\` configured |`)
|
|
186
205
|
if (codeReview && !codeReview.skipped) {
|
|
187
206
|
const codeStatus = codeReview.ok ? '✅ Passed' : '❌ Failed'
|
|
188
|
-
lines.push(
|
|
207
|
+
lines.push(
|
|
208
|
+
`| Code review | ${codeStatus} | ${codeReview.findings.length} findings (${codeReview.model}) |`,
|
|
209
|
+
)
|
|
189
210
|
} else {
|
|
190
211
|
lines.push(`| Code review | ⚪ Skipped | ${codeReview?.summary ?? 'no report'} |`)
|
|
191
212
|
}
|
|
@@ -193,43 +214,7 @@ function renderBody(report, codeReview, runUrl, ok) {
|
|
|
193
214
|
lines.push('</details>')
|
|
194
215
|
lines.push('')
|
|
195
216
|
|
|
196
|
-
|
|
197
|
-
lines.push('<details>')
|
|
198
|
-
lines.push('<summary>🧠 Code review</summary>')
|
|
199
|
-
lines.push('')
|
|
200
|
-
lines.push(`**Verdict:** ${codeReview.verdict} · ${codeReview.model} · ${codeReview.tokens}tok ${formatUsd(codeReview.visionCostUsd)}`)
|
|
201
|
-
if (Array.isArray(codeReview.probes) && codeReview.probes.length > 0) {
|
|
202
|
-
const reproduced = codeReview.probes.filter((p) => p.outcome === 'reproduced').length
|
|
203
|
-
lines.push(` · 🧪 ${codeReview.probes.length} probe${codeReview.probes.length === 1 ? '' : 's'} run, ${reproduced} reproduced`)
|
|
204
|
-
}
|
|
205
|
-
if (typeof codeReview.probeLaneSkipped === 'string') {
|
|
206
|
-
lines.push(` · 🧪 probe lane skipped — ${cell(codeReview.probeLaneSkipped)}`)
|
|
207
|
-
}
|
|
208
|
-
lines.push('')
|
|
209
|
-
lines.push(codeReview.summary)
|
|
210
|
-
lines.push('')
|
|
211
|
-
if (codeReview.findings.length > 0) {
|
|
212
|
-
const evidenceIcon = { exercised: '✅', corroborated: '🔴', not_exercised: '⚪', inconclusive: '❔', reproduced: '🧪' }
|
|
213
|
-
lines.push('| File | Severity | Evidence | Finding |')
|
|
214
|
-
lines.push('| --- | --- | --- | --- |')
|
|
215
|
-
// Findings/evidence strings are model- and probe-emitted — sanitize
|
|
216
|
-
// for the markdown table and bound the section so an oversized report
|
|
217
|
-
// can't push the body past GitHub's 65536-char comment limit.
|
|
218
|
-
const MAX_FINDING_ROWS = 25
|
|
219
|
-
for (const f of codeReview.findings.slice(0, MAX_FINDING_ROWS)) {
|
|
220
|
-
const ev = f.evidence
|
|
221
|
-
? `${evidenceIcon[f.evidence.status] ?? '❔'} ${cell(f.evidence.detail)}`
|
|
222
|
-
: '—'
|
|
223
|
-
lines.push(`| \`${cell(f.file)}\` | ${cell(f.severity)} | ${ev} | ${cell(f.message)} |`)
|
|
224
|
-
}
|
|
225
|
-
if (codeReview.findings.length > MAX_FINDING_ROWS) {
|
|
226
|
-
lines.push(`| … | — | — | ${codeReview.findings.length - MAX_FINDING_ROWS} more findings in \`code-review.json\` |`)
|
|
227
|
-
}
|
|
228
|
-
lines.push('')
|
|
229
|
-
}
|
|
230
|
-
lines.push('</details>')
|
|
231
|
-
lines.push('')
|
|
232
|
-
}
|
|
217
|
+
pushCodeReviewDetails(lines, codeReview, inlinePlan)
|
|
233
218
|
|
|
234
219
|
lines.push('<details>')
|
|
235
220
|
lines.push('<summary>✨ Actions</summary>')
|
|
@@ -247,50 +232,182 @@ function renderBody(report, codeReview, runUrl, ok) {
|
|
|
247
232
|
return lines.join('\n')
|
|
248
233
|
}
|
|
249
234
|
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
process.env.ARGUS_REPORT_DIR || 'argus-reviewer-report',
|
|
235
|
+
// The 🧠 Code review details block — shared by the full body and the
|
|
236
|
+
// review-only body (run lane disabled).
|
|
237
|
+
function pushCodeReviewDetails(lines, codeReview, inlinePlan) {
|
|
238
|
+
if (!codeReview || codeReview.skipped) return
|
|
239
|
+
lines.push('<details>')
|
|
240
|
+
lines.push('<summary>🧠 Code review</summary>')
|
|
241
|
+
lines.push('')
|
|
242
|
+
lines.push(
|
|
243
|
+
`**Verdict:** ${codeReview.verdict} · ${codeReview.model} · ${codeReview.tokens}tok ${formatUsd(codeReview.visionCostUsd)}`,
|
|
260
244
|
)
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
245
|
+
// U7 triage record — Jev annotate/route signals, never the gate.
|
|
246
|
+
if (codeReview.triage) {
|
|
247
|
+
const t = codeReview.triage
|
|
248
|
+
if (t.unadjudicated === true) {
|
|
249
|
+
lines.push(` · 🧭 triage unadjudicated — Jev unavailable`)
|
|
250
|
+
} else {
|
|
251
|
+
lines.push(
|
|
252
|
+
` · 🧭 triage: risk ${t.risk ?? '?'}/5` +
|
|
253
|
+
`${typeof t.needsDeepReview === 'number' ? ` · deep-review ${t.needsDeepReview.toFixed(2)}` : ''}` +
|
|
254
|
+
`${t.topRiskArea !== undefined ? ` · top area \`${cell(t.topRiskArea)}\`` : ''}` +
|
|
255
|
+
` (${cell(t.mode)})`,
|
|
256
|
+
)
|
|
271
257
|
}
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
276
|
-
codeReview
|
|
258
|
+
}
|
|
259
|
+
if (Array.isArray(codeReview.probes) && codeReview.probes.length > 0) {
|
|
260
|
+
const reproduced = codeReview.probes.filter((p) => p.outcome === 'reproduced').length
|
|
261
|
+
lines.push(
|
|
262
|
+
` · 🧪 ${codeReview.probes.length} probe${codeReview.probes.length === 1 ? '' : 's'} run, ${reproduced} reproduced`,
|
|
263
|
+
)
|
|
264
|
+
}
|
|
265
|
+
if (typeof codeReview.probeLaneSkipped === 'string') {
|
|
266
|
+
lines.push(` · 🧪 probe lane skipped — ${cell(codeReview.probeLaneSkipped)}`)
|
|
267
|
+
}
|
|
268
|
+
lines.push('')
|
|
269
|
+
lines.push(codeReview.summary)
|
|
270
|
+
lines.push('')
|
|
271
|
+
if (codeReview.findings.length > 0) {
|
|
272
|
+
const evidenceIcon = {
|
|
273
|
+
exercised: '✅',
|
|
274
|
+
corroborated: '🔴',
|
|
275
|
+
not_exercised: '⚪',
|
|
276
|
+
inconclusive: '❔',
|
|
277
|
+
reproduced: '🧪',
|
|
278
|
+
}
|
|
279
|
+
lines.push('| File | Severity | p | Category | Evidence | Finding |')
|
|
280
|
+
lines.push('| --- | --- | --- | --- | --- | --- |')
|
|
281
|
+
// Findings/evidence strings are model- and probe-emitted — sanitize
|
|
282
|
+
// for the markdown table and bound the section so an oversized report
|
|
283
|
+
// can't push the body past GitHub's 65536-char comment limit.
|
|
284
|
+
const MAX_FINDING_ROWS = 25
|
|
285
|
+
for (const f of codeReview.findings.slice(0, MAX_FINDING_ROWS)) {
|
|
286
|
+
const ev = f.evidence
|
|
287
|
+
? `${evidenceIcon[f.evidence.status] ?? '❔'} ${cell(f.evidence.detail)}`
|
|
288
|
+
: '—'
|
|
289
|
+
// U8 — Jev P(true positive); unadjudicated findings render '—'.
|
|
290
|
+
const p = typeof f.p === 'number' ? f.p.toFixed(2) : '—'
|
|
291
|
+
lines.push(
|
|
292
|
+
`| \`${cell(f.file)}\` | ${cell(f.severity)} | ${p} | ${cell(f.category ?? '—')} | ${ev} | ${cell(f.message)} |`,
|
|
293
|
+
)
|
|
294
|
+
}
|
|
295
|
+
if (codeReview.findings.length > MAX_FINDING_ROWS) {
|
|
296
|
+
lines.push(
|
|
297
|
+
`| … | — | — | — | — | ${codeReview.findings.length - MAX_FINDING_ROWS} more findings in \`code-review.json\` |`,
|
|
298
|
+
)
|
|
299
|
+
}
|
|
300
|
+
lines.push('')
|
|
301
|
+
// Inline-comment cap note — dedup'd fresh findings the
|
|
302
|
+
// review.maxComments budget didn't post (TCA max_comments).
|
|
303
|
+
if (inlinePlan !== undefined && inlinePlan.dropped > 0) {
|
|
304
|
+
lines.push(
|
|
305
|
+
`*+${inlinePlan.dropped} inline-eligible finding(s) not posted — \`review.maxComments\` cap ${inlinePlan.cap}.*`,
|
|
306
|
+
)
|
|
307
|
+
lines.push('')
|
|
308
|
+
}
|
|
309
|
+
// Secrets-lane audit line — adjudicated/suppressed counts, never literals.
|
|
310
|
+
if (codeReview.secretsScan) {
|
|
311
|
+
if (typeof codeReview.secretsScan.skipped === 'string') {
|
|
312
|
+
lines.push(`*🔐 secrets scan skipped — ${cell(codeReview.secretsScan.skipped)}*`)
|
|
313
|
+
} else if (Array.isArray(codeReview.secretsScan.records)) {
|
|
314
|
+
const suppressed = codeReview.secretsScan.records.filter((r) => r.suppressed).length
|
|
315
|
+
const unadj = codeReview.secretsScan.records.filter((r) => !r.adjudicated).length
|
|
316
|
+
lines.push(
|
|
317
|
+
`*🔐 secrets scan: ${codeReview.secretsScan.records.length} candidate(s)` +
|
|
318
|
+
`${suppressed > 0 ? `, ${suppressed} adjudicated-suppressed` : ''}` +
|
|
319
|
+
`${unadj > 0 ? `, ${unadj} unadjudicated` : ''}` +
|
|
320
|
+
`${codeReview.secretsScan.overflow > 0 ? `, +${codeReview.secretsScan.overflow} over cap` : ''}.*`,
|
|
321
|
+
)
|
|
322
|
+
}
|
|
323
|
+
lines.push('')
|
|
277
324
|
}
|
|
278
325
|
}
|
|
326
|
+
// U8 adjudication audit — outside the findings guard so suppressed-
|
|
327
|
+
// only reviews still show what Jev removed. p values live on the
|
|
328
|
+
// findings table and in code-review.json records.
|
|
329
|
+
if (codeReview.findingAdjudication && Array.isArray(codeReview.findingAdjudication.records)) {
|
|
330
|
+
const fa = codeReview.findingAdjudication
|
|
331
|
+
if (fa.unadjudicated === true) {
|
|
332
|
+
lines.push('*🧮 adjudication: unadjudicated — Jev unavailable, nothing suppressed.*')
|
|
333
|
+
} else {
|
|
334
|
+
const suppressed = fa.records.filter((r) => r.suppressed).length
|
|
335
|
+
const unadj = fa.records.filter((r) => !r.adjudicated).length
|
|
336
|
+
lines.push(
|
|
337
|
+
`*🧮 adjudication: ${fa.records.length} finding(s) scored` +
|
|
338
|
+
`${suppressed > 0 ? `, ${suppressed} suppressed (nit/q)` : ''}` +
|
|
339
|
+
`${unadj > 0 ? `, ${unadj} unadjudicated` : ''}` +
|
|
340
|
+
`${fa.overflow > 0 ? `, +${fa.overflow} over cap` : ''}.*`,
|
|
341
|
+
)
|
|
342
|
+
}
|
|
343
|
+
lines.push('')
|
|
344
|
+
}
|
|
345
|
+
lines.push('</details>')
|
|
346
|
+
lines.push('')
|
|
347
|
+
}
|
|
279
348
|
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
const
|
|
284
|
-
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
349
|
+
// Sticky body for `run: 'false'` consumers — no run.json exists by
|
|
350
|
+
// design, so the body and conclusion reflect code-review alone.
|
|
351
|
+
function renderReviewOnlyBody(codeReview, runUrl, ok, inlinePlan) {
|
|
352
|
+
const lines = []
|
|
353
|
+
lines.push(SENTINEL)
|
|
354
|
+
lines.push('')
|
|
355
|
+
lines.push(`## argus-reviewer ${ok ? '✅ PASS' : '❌ FAIL'}`)
|
|
356
|
+
lines.push('')
|
|
357
|
+
if (!codeReview) {
|
|
358
|
+
lines.push(
|
|
359
|
+
'**Summary:** code-review only (run lane disabled) — no `code-review.json` found. The review step crashed or produced no report; the commit status fails closed — check the action logs before merging.',
|
|
360
|
+
)
|
|
361
|
+
} else if (codeReview.skipped) {
|
|
362
|
+
lines.push(
|
|
363
|
+
`**Summary:** code-review only (run lane disabled) — review skipped: ${cell(codeReview.summary)}`,
|
|
364
|
+
)
|
|
365
|
+
} else {
|
|
366
|
+
lines.push(
|
|
367
|
+
`**Summary:** code review only (run lane disabled) · verdict **${codeReview.verdict}** · ` +
|
|
368
|
+
`${codeReview.findings.length} finding(s) · ${codeReview.model} · ` +
|
|
369
|
+
`${codeReview.tokens}tok ${formatUsd(codeReview.visionCostUsd)}`,
|
|
370
|
+
)
|
|
371
|
+
}
|
|
372
|
+
lines.push('')
|
|
373
|
+
lines.push('<details>')
|
|
374
|
+
lines.push('<summary>🚥 Pre-merge checks</summary>')
|
|
375
|
+
lines.push('')
|
|
376
|
+
lines.push('| Check | Status | Explanation |')
|
|
377
|
+
lines.push('| --- | --- | --- |')
|
|
378
|
+
lines.push('| OpenRouter key | ✅ Passed | `OPENROUTER_API_KEY` configured |')
|
|
379
|
+
if (codeReview && !codeReview.skipped) {
|
|
380
|
+
const codeStatus = codeReview.ok ? '✅ Passed' : '❌ Failed'
|
|
381
|
+
lines.push(
|
|
382
|
+
`| Code review | ${codeStatus} | ${codeReview.findings.length} findings (${codeReview.model}) |`,
|
|
383
|
+
)
|
|
384
|
+
} else {
|
|
385
|
+
lines.push(
|
|
386
|
+
`| Code review | ${codeReview ? '⚪ Skipped' : '❌ Failed'} | ${cell(codeReview?.summary ?? 'no report')} |`,
|
|
387
|
+
)
|
|
388
|
+
}
|
|
389
|
+
lines.push('')
|
|
390
|
+
lines.push('</details>')
|
|
391
|
+
lines.push('')
|
|
392
|
+
pushCodeReviewDetails(lines, codeReview, inlinePlan)
|
|
393
|
+
if (runUrl) lines.push(`[View run](${runUrl})`)
|
|
394
|
+
lines.push('')
|
|
395
|
+
lines.push('<details>')
|
|
396
|
+
lines.push('<summary>✨ Actions</summary>')
|
|
397
|
+
lines.push('')
|
|
398
|
+
lines.push('- [ ] Re-run argus-reviewer')
|
|
399
|
+
lines.push('')
|
|
400
|
+
lines.push('</details>')
|
|
401
|
+
lines.push('')
|
|
402
|
+
lines.push('---')
|
|
403
|
+
lines.push('')
|
|
404
|
+
lines.push('<sub>`argus-reviewer` — self-hosted, BYOK OpenRouter UI regression.</sub>')
|
|
405
|
+
lines.push('')
|
|
406
|
+
return lines.join('\n')
|
|
407
|
+
}
|
|
291
408
|
|
|
292
|
-
async function
|
|
293
|
-
if (!pr || !codeReview || codeReview.skipped || !codeReview.findings) return
|
|
409
|
+
async function planInlineComments(pr, codeReview) {
|
|
410
|
+
if (!pr || !codeReview || codeReview.skipped || !codeReview.findings) return undefined
|
|
294
411
|
// Must match the severity vocabulary emitted by the code-review schema
|
|
295
412
|
// (src/cli.ts): bug/risk are inline-worthy; nit/q stay in the sticky body.
|
|
296
413
|
const inlineSeverities = ['bug', 'risk']
|
|
@@ -301,6 +418,8 @@ async function postInlineComments(pr, codeReview) {
|
|
|
301
418
|
line: f.line,
|
|
302
419
|
side: 'RIGHT',
|
|
303
420
|
body: `**argus-reviewer ${f.severity}:** ${f.message}${
|
|
421
|
+
f.category ? ` \`${f.category}\`` : ''
|
|
422
|
+
}${
|
|
304
423
|
f.evidence && f.evidence.status === 'reproduced'
|
|
305
424
|
? '\n\n*🧪 Reproduced by an Argus probe — fails on this PR head, clean on base. See workflow artifacts.*'
|
|
306
425
|
: f.evidence && f.evidence.status !== 'exercised'
|
|
@@ -308,7 +427,7 @@ async function postInlineComments(pr, codeReview) {
|
|
|
308
427
|
: ''
|
|
309
428
|
}`,
|
|
310
429
|
}))
|
|
311
|
-
if (comments.length === 0) return
|
|
430
|
+
if (comments.length === 0) return { capped: [], dropped: 0, cap: 0 }
|
|
312
431
|
|
|
313
432
|
// Re-runs on the same SHA must not duplicate inline comments — the sticky
|
|
314
433
|
// body is upserted but review comments are not. Paginate fully (100/page)
|
|
@@ -336,8 +455,17 @@ async function postInlineComments(pr, codeReview) {
|
|
|
336
455
|
page += 1
|
|
337
456
|
}
|
|
338
457
|
const fresh = comments.filter((c) => !posted.has(dedupKey(c.path, c.line, c.body)))
|
|
339
|
-
if (fresh.length === 0) return
|
|
340
458
|
|
|
459
|
+
// review.maxComments caps inline noise (TCA max_comments) — applied
|
|
460
|
+
// after dedup so already-posted comments don't eat the budget. A cap of
|
|
461
|
+
// 0 disables inline posting entirely (no empty review).
|
|
462
|
+
const cap = typeof codeReview.maxComments === 'number' ? codeReview.maxComments : 20
|
|
463
|
+
const capped = fresh.slice(0, cap)
|
|
464
|
+
return { capped, dropped: fresh.length - capped.length, cap }
|
|
465
|
+
}
|
|
466
|
+
|
|
467
|
+
async function postInlineComments(pr, plan) {
|
|
468
|
+
if (plan === undefined || plan.capped.length === 0) return
|
|
341
469
|
// One batched review instead of N createReviewComment calls — avoids
|
|
342
470
|
// secondary rate limits on large findings sets.
|
|
343
471
|
try {
|
|
@@ -347,13 +475,65 @@ async function postInlineComments(pr, codeReview) {
|
|
|
347
475
|
pull_number: pr.number,
|
|
348
476
|
commit_id: pr.head.sha,
|
|
349
477
|
event: 'COMMENT',
|
|
350
|
-
comments:
|
|
478
|
+
comments: plan.capped,
|
|
351
479
|
})
|
|
352
480
|
} catch (e) {
|
|
353
481
|
core.warning(`inline review failed: ${e.message}`)
|
|
354
482
|
}
|
|
355
483
|
}
|
|
356
484
|
|
|
485
|
+
async function main() {
|
|
486
|
+
const pr = context.payload && context.payload.pull_request
|
|
487
|
+
const owner = context.repo.owner
|
|
488
|
+
const repo = context.repo.repo
|
|
489
|
+
const hasKey = !!process.env.OPENROUTER_API_KEY
|
|
490
|
+
const workDir = process.env.VISION_E2E_WORKING_DIR || ''
|
|
491
|
+
const reportDir = path.resolve(
|
|
492
|
+
process.env.GITHUB_WORKSPACE,
|
|
493
|
+
workDir,
|
|
494
|
+
process.env.ARGUS_REPORT_DIR || 'argus-reviewer-report',
|
|
495
|
+
)
|
|
496
|
+
const runUrl = `${process.env.GITHUB_SERVER_URL}/${owner}/${repo}/actions/runs/${process.env.GITHUB_RUN_ID}`
|
|
497
|
+
|
|
498
|
+
let report
|
|
499
|
+
let codeReview
|
|
500
|
+
if (hasKey) {
|
|
501
|
+
try {
|
|
502
|
+
const raw = fs.readFileSync(path.join(reportDir, 'run.json'), 'utf8')
|
|
503
|
+
report = JSON.parse(raw)
|
|
504
|
+
} catch {
|
|
505
|
+
report = undefined
|
|
506
|
+
}
|
|
507
|
+
try {
|
|
508
|
+
const raw = fs.readFileSync(path.join(reportDir, 'code-review.json'), 'utf8')
|
|
509
|
+
codeReview = JSON.parse(raw)
|
|
510
|
+
} catch {
|
|
511
|
+
codeReview = undefined
|
|
512
|
+
}
|
|
513
|
+
}
|
|
514
|
+
|
|
515
|
+
// Missing code-review.json after a continue-on-error step means the review
|
|
516
|
+
// crashed, not that it skipped — an intentional skip writes ok+skipped.
|
|
517
|
+
// Fail closed rather than reporting it as a clean skip.
|
|
518
|
+
const codeReviewOk = codeReview != null && codeReview.ok === true
|
|
519
|
+
// run: 'false' consumers have no run.json by design — the conclusion then
|
|
520
|
+
// reflects the code-review verdict alone.
|
|
521
|
+
const runDisabled = process.env.ARGUS_RUN_DISABLED === '1'
|
|
522
|
+
const ok = (runDisabled || report?.ok === true) && codeReviewOk
|
|
523
|
+
const conclusion = !hasKey ? 'neutral' : ok ? 'success' : 'failure'
|
|
524
|
+
const inlinePlan = hasKey ? await planInlineComments(pr, codeReview) : undefined
|
|
525
|
+
const body = !hasKey
|
|
526
|
+
? renderMissingKeyBody()
|
|
527
|
+
: runDisabled
|
|
528
|
+
? renderReviewOnlyBody(codeReview, runUrl, ok, inlinePlan)
|
|
529
|
+
: report === undefined
|
|
530
|
+
? renderNoReportBody(reportDir, runUrl)
|
|
531
|
+
: renderBody(report, codeReview, runUrl, ok, inlinePlan)
|
|
532
|
+
|
|
533
|
+
// Eligibility + dedup for inline comments, computed before the sticky
|
|
534
|
+
// body renders so the "+N not posted" note counts the *fresh* set — the
|
|
535
|
+
// cap is applied to fresh, not to raw findings (already-posted comments
|
|
536
|
+
// must not inflate the dropped count).
|
|
357
537
|
if (pr) {
|
|
358
538
|
const { data: comments } = await github.rest.issues.listComments({
|
|
359
539
|
owner,
|
|
@@ -377,7 +557,7 @@ async function postInlineComments(pr, codeReview) {
|
|
|
377
557
|
body,
|
|
378
558
|
})
|
|
379
559
|
}
|
|
380
|
-
await postInlineComments(pr,
|
|
560
|
+
await postInlineComments(pr, inlinePlan)
|
|
381
561
|
}
|
|
382
562
|
|
|
383
563
|
const sha = pr ? pr.head.sha : context.sha
|
package/dist/api.d.ts
CHANGED
|
@@ -7,6 +7,7 @@ import { Config, defineConfig } from './config.js';
|
|
|
7
7
|
import { ErrorRecord } from './journal/schema.js';
|
|
8
8
|
import { Logger } from './log.js';
|
|
9
9
|
export { defineConfig };
|
|
10
|
+
export { DecisionClient, DecisionError, JEV_DEFAULT_MODEL, type DecisionAnswer, type DecisionClientOptions, type DecisionErrorKind, type DecisionQuestion, } from './vision/decisions.js';
|
|
10
11
|
/**
|
|
11
12
|
* Test-facing API (R13). Test files are plain TypeScript using a `td` object:
|
|
12
13
|
*
|
package/dist/api.js
CHANGED
|
@@ -4,6 +4,7 @@ import { loadFlow, saveFlow } from './cache/store.js';
|
|
|
4
4
|
import { Ledger } from './vision/ledger.js';
|
|
5
5
|
import { defineConfig } from './config.js';
|
|
6
6
|
export { defineConfig };
|
|
7
|
+
export { DecisionClient, DecisionError, JEV_DEFAULT_MODEL, } from './vision/decisions.js';
|
|
7
8
|
const KEY_ALIASES = {
|
|
8
9
|
tab: 'Tab',
|
|
9
10
|
enter: 'Enter',
|