argus-reviewer-e2e 0.1.3 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. package/README.md +86 -49
  2. package/action/action.yml +155 -61
  3. package/action/approval-review.mjs +393 -0
  4. package/action/bootstrap.mjs +102 -0
  5. package/action/cli.mjs +36 -0
  6. package/action/emit-review.mjs +336 -0
  7. package/action/runtime.mjs +154 -0
  8. package/action/sticky-comment.cjs +870 -0
  9. package/dist/api.d.ts +2 -0
  10. package/dist/api.js +4 -0
  11. package/dist/cache/store.d.ts +2 -0
  12. package/dist/cache/store.js +10 -3
  13. package/dist/cli.d.ts +137 -0
  14. package/dist/cli.js +677 -60
  15. package/dist/config.d.ts +67 -2
  16. package/dist/config.js +138 -7
  17. package/dist/debug.d.ts +1 -0
  18. package/dist/debug.js +9 -3
  19. package/dist/engine/loop.d.ts +12 -0
  20. package/dist/engine/loop.js +55 -6
  21. package/dist/evidence/ci.d.ts +3 -0
  22. package/dist/evidence/ci.js +3 -1
  23. package/dist/pipeline/budget.d.ts +11 -0
  24. package/dist/pipeline/budget.js +48 -0
  25. package/dist/pipeline/contracts.d.ts +12 -0
  26. package/dist/pipeline/contracts.js +16 -0
  27. package/dist/pipeline/verify.d.ts +27 -0
  28. package/dist/pipeline/verify.js +197 -0
  29. package/dist/probe/queue.d.ts +4 -1
  30. package/dist/probe/queue.js +46 -8
  31. package/dist/report/manifest.d.ts +87 -0
  32. package/dist/report/manifest.js +143 -0
  33. package/dist/report/run.d.ts +9 -0
  34. package/dist/report/run.js +6 -0
  35. package/dist/review/adjudicate.d.ts +63 -0
  36. package/dist/review/adjudicate.js +111 -0
  37. package/dist/review/secrets.d.ts +90 -0
  38. package/dist/review/secrets.js +224 -0
  39. package/dist/review/triage.d.ts +76 -0
  40. package/dist/review/triage.js +163 -0
  41. package/dist/trust.d.ts +50 -0
  42. package/dist/trust.js +103 -0
  43. package/dist/vision/cost.d.ts +14 -1
  44. package/dist/vision/cost.js +13 -0
  45. package/dist/vision/decisions.d.ts +95 -0
  46. package/dist/vision/decisions.js +232 -0
  47. package/package.json +9 -8
  48. package/action/sticky-comment.mjs +0 -404
@@ -1,404 +0,0 @@
1
- /* global github, context, core, require, process */
2
-
3
- const fs = require('fs')
4
- const path = require('path')
5
-
6
- const SENTINEL = '<!-- argus-reviewer -->'
7
-
8
- function formatUsd(n) {
9
- return `$${(n || 0).toFixed(6)}`
10
- }
11
-
12
- /** Escape a report string for one markdown table cell. */
13
- function cell(s) {
14
- return String(s ?? '').replace(/\|/g, '\\|').replace(/[\r\n]+/g, ' ').slice(0, 200)
15
- }
16
-
17
- function renderMissingKeyBody() {
18
- const lines = []
19
- lines.push(SENTINEL)
20
- lines.push('')
21
- lines.push('## argus-reviewer ⚪ skipped')
22
- lines.push('')
23
- lines.push('`OPENROUTER_API_KEY` is not configured. Add it as a repository or workflow secret to run argus-reviewer.')
24
- lines.push('')
25
- lines.push('This status is intentionally neutral, not a failure.')
26
- lines.push('')
27
- return lines.join('\n')
28
- }
29
-
30
- function renderNoReportBody(reportDir, runUrl) {
31
- const lines = []
32
- lines.push(SENTINEL)
33
- lines.push('')
34
- lines.push('## argus-reviewer ⚠️ no report')
35
- lines.push('')
36
- lines.push(`The run step produced no \`run.json\` under \`${reportDir}\`. The commit status fails closed — check the action logs before merging.`)
37
- lines.push('')
38
- lines.push(`[View run](${runUrl})`)
39
- lines.push('')
40
- return lines.join('\n')
41
- }
42
-
43
- function renderBody(report, codeReview, runUrl, ok) {
44
- if (!report) return renderMissingKeyBody()
45
-
46
- const lines = []
47
- const budgetCap = report.config?.budgetUsd ?? 0
48
- const healCount = report.tests.reduce((n, t) => n + (t.healEvents?.length ?? 0), 0)
49
- const assertCount = report.tests.reduce((n, t) => n + (t.asserts?.length ?? 0), 0)
50
- const assertFails = report.tests.reduce(
51
- (n, t) => n + (t.asserts?.filter((a) => a.verdict === 'fail').length ?? 0),
52
- 0,
53
- )
54
- const trace = report.trace ?? {}
55
-
56
- lines.push(SENTINEL)
57
- lines.push('')
58
- lines.push(`## argus-reviewer ${ok ? '✅ PASS' : '❌ FAIL'}`)
59
- lines.push('')
60
- lines.push(
61
- `**Summary:** ${report.totals.passed}/${report.totals.tests} passed · ` +
62
- `${report.totals.visionCalls} vision calls · ` +
63
- `${formatUsd(report.totals.visionCostUsd)} spend · ` +
64
- `${report.totals.sandboxSeconds.toFixed(1)}s sandbox`,
65
- )
66
- lines.push('')
67
-
68
- lines.push('<details>')
69
- lines.push('<summary>📝 Summary</summary>')
70
- lines.push('')
71
- lines.push('**What ran**')
72
- for (const t of report.tests) {
73
- lines.push(`- \`${path.basename(t.file)}\` — ${t.name}`)
74
- }
75
- lines.push('')
76
- lines.push(`**Risk:** ${ok ? 'Low — UI regression tests and code review passed; no heals or failures.' : 'High — investigate failures before merge.'}`)
77
- lines.push('')
78
- if (Object.keys(trace).length > 0) {
79
- lines.push('**Trace**')
80
- for (const [k, v] of Object.entries(trace)) {
81
- lines.push(`- ${k}: \`${v}\``)
82
- }
83
- lines.push('')
84
- }
85
- lines.push('</details>')
86
- lines.push('')
87
-
88
- lines.push('<details>')
89
- lines.push(`<summary>📒 Tests (${report.totals.tests})</summary>`)
90
- lines.push('')
91
- lines.push('| Test | Result | Calls | Cost | Heals | Asserts |')
92
- lines.push('| --- | --- | ---: | ---: | ---: | ---: |')
93
- for (const t of report.tests) {
94
- const result = t.ok ? '✅ pass' : '❌ fail'
95
- lines.push(`| ${t.name} | ${result} | ${t.visionCalls} | ${formatUsd(t.visionCostUsd)} | ${t.healEvents?.length ?? 0} | ${t.asserts?.length ?? 0} |`)
96
- }
97
- lines.push('')
98
- lines.push('</details>')
99
- lines.push('')
100
-
101
- lines.push('<details>')
102
- lines.push('<summary>💰 Cost ledger</summary>')
103
- lines.push('')
104
- lines.push('| Line item | Value |')
105
- lines.push('| --- | ---: |')
106
- lines.push(`| Vision calls | ${report.totals.visionCalls} |`)
107
- const perCall =
108
- report.totals.visionCalls > 0
109
- ? formatUsd(report.totals.visionCostUsd / report.totals.visionCalls)
110
- : '$0.00'
111
- lines.push(`| Per-call cost (avg) | ${perCall} |`)
112
- for (const model of Object.keys(report.totals.callsByModel ?? {}).sort()) {
113
- lines.push(`| Calls (${model}) | ${report.totals.callsByModel[model]} |`)
114
- lines.push(`| Spend (${model}) | ${formatUsd(report.totals.costByModel?.[model] ?? 0)} |`)
115
- }
116
- lines.push(`| Total vision spend | ${formatUsd(report.totals.visionCostUsd)} |`)
117
- lines.push(`| Sandbox seconds | ${report.totals.sandboxSeconds.toFixed(1)}s |`)
118
- if (budgetCap > 0) {
119
- lines.push(`| Budget cap | ${formatUsd(budgetCap)} |`)
120
- lines.push(`| Budget exceeded | ${report.totals.budgetExceeded ? '⚠️ yes' : '✅ no'} |`)
121
- }
122
- lines.push('')
123
- lines.push('</details>')
124
- lines.push('')
125
-
126
- lines.push('<details>')
127
- lines.push('<summary>🔧 Heal events</summary>')
128
- lines.push('')
129
- const heals = report.tests.flatMap((t) => t.healEvents ?? [])
130
- if (heals.length === 0) {
131
- lines.push('No heals this run.')
132
- } else {
133
- for (const h of heals) {
134
- lines.push(`- \`${h.instruction}\` healed with ${h.model || 'unknown model'}`)
135
- }
136
- }
137
- lines.push('')
138
- lines.push('</details>')
139
- lines.push('')
140
-
141
- lines.push('<details>')
142
- lines.push('<summary>✅ Assertions</summary>')
143
- lines.push('')
144
- let any = false
145
- for (const t of report.tests) {
146
- if (!t.asserts || t.asserts.length === 0) continue
147
- any = true
148
- lines.push(`**${t.name}**`)
149
- for (const a of t.asserts) {
150
- const icon = a.verdict === 'pass' ? '✅' : a.verdict === 'fail' ? '❌' : '⚪'
151
- lines.push(`- ${icon} *${a.question}* — ${a.reasoning}`)
152
- }
153
- lines.push('')
154
- }
155
- if (!any) {
156
- lines.push('No assertions recorded.')
157
- lines.push('')
158
- }
159
- lines.push('</details>')
160
- lines.push('')
161
-
162
- lines.push('<details>')
163
- lines.push('<summary>📂 Evidence</summary>')
164
- lines.push('')
165
- if (report.artifacts && report.artifacts.videos.length > 0) {
166
- for (const v of report.artifacts.videos) lines.push(`- video: \`${v}\``)
167
- }
168
- if (runUrl) lines.push(`- [workflow run / artifacts](${runUrl})`)
169
- if ((!report.artifacts || report.artifacts.videos.length === 0) && !runUrl) {
170
- lines.push('No artifact links available.')
171
- }
172
- lines.push('')
173
- lines.push('</details>')
174
- lines.push('')
175
-
176
- lines.push('<details>')
177
- lines.push('<summary>🚥 Pre-merge checks</summary>')
178
- lines.push('')
179
- lines.push('| Check | Status | Explanation |')
180
- lines.push('| --- | --- | --- |')
181
- lines.push(`| Tests | ${report.ok ? '✅ Passed' : '❌ Failed'} | ${report.totals.passed}/${report.totals.tests} tests passed |`)
182
- lines.push(`| Budget | ${report.totals.budgetExceeded ? '⚠️ Warning' : '✅ Passed'} | ${formatUsd(report.totals.visionCostUsd)} spent${budgetCap > 0 ? ` of ${formatUsd(budgetCap)}` : ''} |`)
183
- lines.push(`| Heal events | ${healCount === 0 ? '✅ Passed' : '⚠️ Warning'} | ${healCount} heal event${healCount === 1 ? '' : 's'} |`)
184
- lines.push(`| Assertions | ${assertFails === 0 ? '✅ Passed' : '❌ Failed'} | ${assertFails === 0 ? assertCount : `${assertFails} failed`} assertion${assertCount === 1 ? '' : 's'} |`)
185
- lines.push(`| OpenRouter key | ✅ Passed | \`OPENROUTER_API_KEY\` configured |`)
186
- if (codeReview && !codeReview.skipped) {
187
- const codeStatus = codeReview.ok ? '✅ Passed' : '❌ Failed'
188
- lines.push(`| Code review | ${codeStatus} | ${codeReview.findings.length} findings (${codeReview.model}) |`)
189
- } else {
190
- lines.push(`| Code review | ⚪ Skipped | ${codeReview?.summary ?? 'no report'} |`)
191
- }
192
- lines.push('')
193
- lines.push('</details>')
194
- lines.push('')
195
-
196
- if (codeReview && !codeReview.skipped) {
197
- lines.push('<details>')
198
- lines.push('<summary>🧠 Code review</summary>')
199
- lines.push('')
200
- lines.push(`**Verdict:** ${codeReview.verdict} · ${codeReview.model} · ${codeReview.tokens}tok ${formatUsd(codeReview.visionCostUsd)}`)
201
- if (Array.isArray(codeReview.probes) && codeReview.probes.length > 0) {
202
- const reproduced = codeReview.probes.filter((p) => p.outcome === 'reproduced').length
203
- lines.push(` · 🧪 ${codeReview.probes.length} probe${codeReview.probes.length === 1 ? '' : 's'} run, ${reproduced} reproduced`)
204
- }
205
- if (typeof codeReview.probeLaneSkipped === 'string') {
206
- lines.push(` · 🧪 probe lane skipped — ${cell(codeReview.probeLaneSkipped)}`)
207
- }
208
- lines.push('')
209
- lines.push(codeReview.summary)
210
- lines.push('')
211
- if (codeReview.findings.length > 0) {
212
- const evidenceIcon = { exercised: '✅', corroborated: '🔴', not_exercised: '⚪', inconclusive: '❔', reproduced: '🧪' }
213
- lines.push('| File | Severity | Evidence | Finding |')
214
- lines.push('| --- | --- | --- | --- |')
215
- // Findings/evidence strings are model- and probe-emitted — sanitize
216
- // for the markdown table and bound the section so an oversized report
217
- // can't push the body past GitHub's 65536-char comment limit.
218
- const MAX_FINDING_ROWS = 25
219
- for (const f of codeReview.findings.slice(0, MAX_FINDING_ROWS)) {
220
- const ev = f.evidence
221
- ? `${evidenceIcon[f.evidence.status] ?? '❔'} ${cell(f.evidence.detail)}`
222
- : '—'
223
- lines.push(`| \`${cell(f.file)}\` | ${cell(f.severity)} | ${ev} | ${cell(f.message)} |`)
224
- }
225
- if (codeReview.findings.length > MAX_FINDING_ROWS) {
226
- lines.push(`| … | — | — | ${codeReview.findings.length - MAX_FINDING_ROWS} more findings in \`code-review.json\` |`)
227
- }
228
- lines.push('')
229
- }
230
- lines.push('</details>')
231
- lines.push('')
232
- }
233
-
234
- lines.push('<details>')
235
- lines.push('<summary>✨ Actions</summary>')
236
- lines.push('')
237
- lines.push('- [ ] Re-run argus-reviewer')
238
- lines.push('- [ ] Open a heal PR')
239
- lines.push('- [ ] Record a new flow')
240
- lines.push('')
241
- lines.push('</details>')
242
- lines.push('')
243
- lines.push('---')
244
- lines.push('')
245
- lines.push('<sub>`argus-reviewer` — self-hosted, BYOK OpenRouter UI regression.</sub>')
246
- lines.push('')
247
- return lines.join('\n')
248
- }
249
-
250
- async function main() {
251
- const pr = context.payload && context.payload.pull_request
252
- const owner = context.repo.owner
253
- const repo = context.repo.repo
254
- const hasKey = !!process.env.OPENROUTER_API_KEY
255
- const workDir = process.env.VISION_E2E_WORKING_DIR || ''
256
- const reportDir = path.resolve(
257
- process.env.GITHUB_WORKSPACE,
258
- workDir,
259
- process.env.ARGUS_REPORT_DIR || 'argus-reviewer-report',
260
- )
261
- const runUrl = `${process.env.GITHUB_SERVER_URL}/${owner}/${repo}/actions/runs/${process.env.GITHUB_RUN_ID}`
262
-
263
- let report
264
- let codeReview
265
- if (hasKey) {
266
- try {
267
- const raw = fs.readFileSync(path.join(reportDir, 'run.json'), 'utf8')
268
- report = JSON.parse(raw)
269
- } catch {
270
- report = undefined
271
- }
272
- try {
273
- const raw = fs.readFileSync(path.join(reportDir, 'code-review.json'), 'utf8')
274
- codeReview = JSON.parse(raw)
275
- } catch {
276
- codeReview = undefined
277
- }
278
- }
279
-
280
- // Missing code-review.json after a continue-on-error step means the review
281
- // crashed, not that it skipped — an intentional skip writes ok+skipped.
282
- // Fail closed rather than reporting it as a clean skip.
283
- const codeReviewOk = codeReview != null && codeReview.ok === true
284
- const ok = (report?.ok === true) && codeReviewOk
285
- const conclusion = !hasKey ? 'neutral' : ok ? 'success' : 'failure'
286
- const body = !hasKey
287
- ? renderMissingKeyBody()
288
- : report === undefined
289
- ? renderNoReportBody(reportDir, runUrl)
290
- : renderBody(report, codeReview, runUrl, ok)
291
-
292
- async function postInlineComments(pr, codeReview) {
293
- if (!pr || !codeReview || codeReview.skipped || !codeReview.findings) return
294
- // Must match the severity vocabulary emitted by the code-review schema
295
- // (src/cli.ts): bug/risk are inline-worthy; nit/q stay in the sticky body.
296
- const inlineSeverities = ['bug', 'risk']
297
- const comments = codeReview.findings
298
- .filter((f) => f.file && typeof f.line === 'number' && inlineSeverities.includes(f.severity))
299
- .map((f) => ({
300
- path: f.file,
301
- line: f.line,
302
- side: 'RIGHT',
303
- body: `**argus-reviewer ${f.severity}:** ${f.message}${
304
- f.evidence && f.evidence.status === 'reproduced'
305
- ? '\n\n*🧪 Reproduced by an Argus probe — fails on this PR head, clean on base. See workflow artifacts.*'
306
- : f.evidence && f.evidence.status !== 'exercised'
307
- ? `\n\n*CI evidence: ${f.evidence.detail}*`
308
- : ''
309
- }`,
310
- }))
311
- if (comments.length === 0) return
312
-
313
- // Re-runs on the same SHA must not duplicate inline comments — the sticky
314
- // body is upserted but review comments are not. Paginate fully (100/page)
315
- // and scope dedup to the current head: comments on older commits must not
316
- // suppress findings that still apply to this head. The key is the body's
317
- // first line (the finding itself) — trailing evidence notes like the
318
- // `reproduced` upgrade change the body but must not re-post a duplicate.
319
- const dedupKey = (path, line, body) => `${path}:${line}:${body.split('\n')[0]}`
320
- const posted = new Set()
321
- let page = 1
322
- for (;;) {
323
- const { data: existing } = await github.rest.pulls.listReviewComments({
324
- owner: context.repo.owner,
325
- repo: context.repo.repo,
326
- pull_number: pr.number,
327
- per_page: 100,
328
- page,
329
- })
330
- for (const c of existing) {
331
- if (c.body && c.body.startsWith('**argus-reviewer') && c.commit_id === pr.head.sha) {
332
- posted.add(dedupKey(c.path, c.line, c.body))
333
- }
334
- }
335
- if (existing.length < 100) break
336
- page += 1
337
- }
338
- const fresh = comments.filter((c) => !posted.has(dedupKey(c.path, c.line, c.body)))
339
- if (fresh.length === 0) return
340
-
341
- // One batched review instead of N createReviewComment calls — avoids
342
- // secondary rate limits on large findings sets.
343
- try {
344
- await github.rest.pulls.createReview({
345
- owner: context.repo.owner,
346
- repo: context.repo.repo,
347
- pull_number: pr.number,
348
- commit_id: pr.head.sha,
349
- event: 'COMMENT',
350
- comments: fresh,
351
- })
352
- } catch (e) {
353
- core.warning(`inline review failed: ${e.message}`)
354
- }
355
- }
356
-
357
- if (pr) {
358
- const { data: comments } = await github.rest.issues.listComments({
359
- owner,
360
- repo,
361
- issue_number: pr.number,
362
- per_page: 100,
363
- })
364
- const existing = comments.find((c) => c.body && c.body.includes(SENTINEL))
365
- if (existing) {
366
- await github.rest.issues.updateComment({
367
- owner,
368
- repo,
369
- comment_id: existing.id,
370
- body,
371
- })
372
- } else {
373
- await github.rest.issues.createComment({
374
- owner,
375
- repo,
376
- issue_number: pr.number,
377
- body,
378
- })
379
- }
380
- await postInlineComments(pr, codeReview)
381
- }
382
-
383
- const sha = pr ? pr.head.sha : context.sha
384
- // Commit statuses have no 'neutral'; a 'pending' skip would wedge a
385
- // required check forever, so skip maps to success with a clear label.
386
- const state = conclusion === 'failure' ? 'failure' : 'success'
387
- const description =
388
- conclusion === 'neutral'
389
- ? 'argus-reviewer skipped (no OPENROUTER_API_KEY)'
390
- : `argus-reviewer ${conclusion}`
391
- await github.rest.repos.createCommitStatus({
392
- owner,
393
- repo,
394
- sha,
395
- state,
396
- description,
397
- context: 'argus-reviewer',
398
- target_url: runUrl,
399
- })
400
-
401
- core.setOutput('conclusion', conclusion)
402
- }
403
-
404
- return await main()