argus-reviewer-e2e 0.1.2 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +30 -5
- package/action/action.yml +22 -0
- package/action/sticky-comment.mjs +287 -79
- package/dist/api.d.ts +1 -0
- package/dist/api.js +1 -0
- package/dist/cli.d.ts +67 -0
- package/dist/cli.js +463 -86
- package/dist/config.d.ts +105 -2
- package/dist/config.js +136 -9
- package/dist/debug.d.ts +1 -0
- package/dist/debug.js +9 -3
- package/dist/detect.d.ts +11 -1
- package/dist/detect.js +19 -3
- package/dist/evidence/ci.d.ts +47 -2
- package/dist/evidence/ci.js +98 -6
- package/dist/evidence/gate.d.ts +10 -0
- package/dist/evidence/gate.js +29 -0
- package/dist/evidence/link.d.ts +6 -1
- package/dist/evidence/link.js +7 -5
- package/dist/executor/sandbox.d.ts +105 -0
- package/dist/executor/sandbox.js +231 -0
- package/dist/probe/author.d.ts +50 -0
- package/dist/probe/author.js +149 -0
- package/dist/probe/harness.d.ts +30 -0
- package/dist/probe/harness.js +99 -0
- package/dist/probe/queue.d.ts +105 -0
- package/dist/probe/queue.js +446 -0
- package/dist/review/adjudicate.d.ts +63 -0
- package/dist/review/adjudicate.js +111 -0
- package/dist/review/secrets.d.ts +88 -0
- package/dist/review/secrets.js +220 -0
- package/dist/review/triage.d.ts +76 -0
- package/dist/review/triage.js +163 -0
- package/dist/trust.d.ts +50 -0
- package/dist/trust.js +103 -0
- package/dist/vision/cost.d.ts +14 -1
- package/dist/vision/cost.js +13 -0
- package/dist/vision/decisions.d.ts +95 -0
- package/dist/vision/decisions.js +232 -0
- package/package.json +3 -2
package/README.md
CHANGED
|
@@ -4,12 +4,20 @@
|
|
|
4
4
|
<img src="docs/assets/social.png" alt="Argus — vision-model E2E testing" width="640" />
|
|
5
5
|
</p>
|
|
6
6
|
|
|
7
|
+
<p align="center">
|
|
8
|
+
<a href="https://www.npmjs.com/package/argus-reviewer-e2e"><img src="https://img.shields.io/npm/v/argus-reviewer-e2e" alt="npm version" /></a>
|
|
9
|
+
<a href="https://github.com/duketopceo/Argus/actions/workflows/ci.yml"><img src="https://github.com/duketopceo/Argus/actions/workflows/ci.yml/badge.svg" alt="CI" /></a>
|
|
10
|
+
<a href="LICENSE"><img src="https://img.shields.io/badge/license-MIT-blue" alt="MIT license" /></a>
|
|
11
|
+
<a href="https://github.com/duketopceo/Argus/security/policy"><img src="https://img.shields.io/badge/security-policy-orange" alt="security policy" /></a>
|
|
12
|
+
</p>
|
|
13
|
+
|
|
7
14
|
Open-source, self-hosted vision-model E2E testing — the hundred-eyed watcher for your UI. Bring your own `OPENROUTER_API_KEY`: record a flow once, fingerprint-cache every step, replay near-free, heal on UI drift, and get results as a check + comment on the GitHub PR.
|
|
8
15
|
|
|
9
16
|
- **Vision-first**: a model looks at a screenshot and decides where to click — no selectors to write or maintain.
|
|
10
17
|
- **Cache-first**: replay costs zero vision calls on an unchanged UI; heals re-spend only on drift and show up as reviewable cache diffs.
|
|
11
18
|
- **Cost-explicit**: every call is metered from OpenRouter's per-call cost and rolled into a per-run dollar figure on the PR.
|
|
12
19
|
- **Grounding specialist**: a `grounding_model` (e.g. a ui-tars-class model) can drive element location with its native coordinate output, verified against the DOM before any click executes.
|
|
20
|
+
- **Execution-backed review**: `code-review` findings carry CI evidence, and with the opt-in sandbox lane (`sandbox: { enabled: true }`) Argus authors a test probe for unexercised findings and runs it in a hardened, network-less Docker container — a finding that fails on head and passes on base is stamped **reproduced**, not just suspected.
|
|
13
21
|
|
|
14
22
|
```bash
|
|
15
23
|
npm i -D argus-reviewer-e2e # or github:duketopceo/argus-reviewer
|
|
@@ -40,7 +48,10 @@ export default {
|
|
|
40
48
|
|
|
41
49
|
The GitHub Action automatically sets `ARGUS_REVIEWER_TRACE` with the repository, PR number, commit, and run id, so every PR review is attributed in OpenRouter without extra config. You can also set `ARGUS_REVIEWER_TRACE` yourself (JSON object) to add more fields.
|
|
42
50
|
|
|
43
|
-
Status: early development. See `action/` for the composite GitHub Action
|
|
51
|
+
Status: early development. See `action/` for the composite GitHub Action,
|
|
52
|
+
`runner/` for self-hosted runner registration, `docs/quickstart.md` for
|
|
53
|
+
setup, `SECURITY.md` for the threat model, and `CONTRIBUTING.md` to hack
|
|
54
|
+
on it.
|
|
44
55
|
|
|
45
56
|
## File structure
|
|
46
57
|
|
|
@@ -52,10 +63,14 @@ argus-reviewer/
|
|
|
52
63
|
├── runner/ # Self-hosted runner registration docs + script
|
|
53
64
|
│ ├── README.md
|
|
54
65
|
│ └── register-runner.sh
|
|
66
|
+
├── electron/ # Local observability dashboard (`npm run app`)
|
|
55
67
|
├── src/
|
|
56
68
|
│ ├── api.ts # Test-facing `test`/`td` API + generated test file renderer
|
|
57
|
-
│ ├── cli.ts #
|
|
69
|
+
│ ├── cli.ts # record · run · code-review · delegate · cache · index · init
|
|
58
70
|
│ ├── config.ts # `argus-reviewer.config.*` loader (legacy `vision-e2e.config.*` accepted)
|
|
71
|
+
│ ├── cache/
|
|
72
|
+
│ │ ├── fingerprint.ts # Per-step screenshot/a11y fingerprint + resolve
|
|
73
|
+
│ │ └── store.ts # Flow cache read/write
|
|
59
74
|
│ ├── driver/
|
|
60
75
|
│ │ ├── browser.ts # Playwright browser launch (chromium/firefox/webkit) + observation capture
|
|
61
76
|
│ │ └── target.ts # Optional local dev-server target process
|
|
@@ -63,9 +78,19 @@ argus-reviewer/
|
|
|
63
78
|
│ │ ├── actions.ts # Low-level page actions (click, type, scroll, …)
|
|
64
79
|
│ │ ├── loop.ts # Vision model record/replay + healing loop
|
|
65
80
|
│ │ └── prompts.ts # OpenRouter action/assertion prompts + JSON schemas
|
|
66
|
-
│ ├──
|
|
67
|
-
│ │ ├──
|
|
68
|
-
│ │
|
|
81
|
+
│ ├── evidence/
|
|
82
|
+
│ │ ├── ci.ts # PR metadata + CI check-run context for findings
|
|
83
|
+
│ │ ├── gate.ts # Fork-PR trust gate (argus-probe label bound to head SHA)
|
|
84
|
+
│ │ └── link.ts # Finding → evidence linkage + comment-safe sanitization
|
|
85
|
+
│ ├── executor/
|
|
86
|
+
│ │ ├── a0.ts # `a0 headless -p` delegation to a user's Agent Zero instance
|
|
87
|
+
│ │ └── sandbox.ts # Hardened Docker runner for generated probes
|
|
88
|
+
│ ├── index/ # Repo index, diff context, cache invalidation
|
|
89
|
+
│ ├── journal/ # Per-run structured journal entries
|
|
90
|
+
│ ├── probe/
|
|
91
|
+
│ │ ├── author.ts # Model-authored regression probe generation + validation
|
|
92
|
+
│ │ ├── harness.ts # vitest/jest/node:test detection + TAP classification
|
|
93
|
+
│ │ └── queue.ts # Head-vs-merge-base probe orchestration
|
|
69
94
|
│ ├── report/
|
|
70
95
|
│ │ ├── comment.ts # Markdown PR comment + commit-status rendering
|
|
71
96
|
│ │ ├── junit.ts # JUnit XML output
|
package/action/action.yml
CHANGED
|
@@ -42,6 +42,18 @@ inputs:
|
|
|
42
42
|
description: Report output dir, passed to both run and code-review so the post step can find run.json/code-review.json regardless of config reportDir
|
|
43
43
|
default: argus-reviewer-report
|
|
44
44
|
required: false
|
|
45
|
+
sandbox:
|
|
46
|
+
description: Enable the B.2 probe lane — authored test probes for unexercised findings run in a hardened Docker container (no network, no secrets, read-only fs). Enable-only — 'false' does not override a config-enabled lane. Requires Docker on the runner; forced off on pull_request_target.
|
|
47
|
+
default: 'false'
|
|
48
|
+
required: false
|
|
49
|
+
max-comments:
|
|
50
|
+
description: Cap on inline review comments posted per run; overflow is summarized in the sticky. Overrides review.maxComments when set.
|
|
51
|
+
default: ''
|
|
52
|
+
required: false
|
|
53
|
+
run:
|
|
54
|
+
description: Run the browser-flow lane (`argus-reviewer run`). Set 'false' for code-review-only consumers — repos without a per-PR web target (CLIs, libraries, infra repos). Skips the Playwright install and run steps; the sticky comment and commit status then reflect code-review alone.
|
|
55
|
+
default: 'true'
|
|
56
|
+
required: false
|
|
45
57
|
outputs:
|
|
46
58
|
conclusion:
|
|
47
59
|
description: 'Run conclusion: success, failure, or neutral'
|
|
@@ -61,6 +73,7 @@ runs:
|
|
|
61
73
|
run: npm ci
|
|
62
74
|
|
|
63
75
|
- name: Install Playwright browser
|
|
76
|
+
if: inputs.run != 'false'
|
|
64
77
|
shell: bash
|
|
65
78
|
working-directory: ${{ inputs.working-directory }}
|
|
66
79
|
env:
|
|
@@ -89,6 +102,10 @@ runs:
|
|
|
89
102
|
if: inputs.index == 'true'
|
|
90
103
|
shell: bash
|
|
91
104
|
working-directory: ${{ inputs.working-directory }}
|
|
105
|
+
env:
|
|
106
|
+
# Token only feeds trust resolution on issue_comment events —
|
|
107
|
+
# pull_request* events read fork status from the event payload.
|
|
108
|
+
GITHUB_TOKEN: ${{ github.token }}
|
|
92
109
|
run: ${{ inputs.cli }} index
|
|
93
110
|
|
|
94
111
|
- name: Run argus-reviewer code review
|
|
@@ -99,6 +116,8 @@ runs:
|
|
|
99
116
|
env:
|
|
100
117
|
OPENROUTER_API_KEY: ${{ inputs.openrouter-api-key }}
|
|
101
118
|
GITHUB_TOKEN: ${{ github.token }}
|
|
119
|
+
ARGUS_SANDBOX: ${{ inputs.sandbox == 'true' && '1' || '' }}
|
|
120
|
+
ARGUS_MAX_COMMENTS: ${{ inputs.max-comments }}
|
|
102
121
|
ARGUS_DEBUG: '1'
|
|
103
122
|
ARGUS_REVIEWER_TRACE: >-
|
|
104
123
|
{"repo":"${{ github.repository }}",
|
|
@@ -111,11 +130,13 @@ runs:
|
|
|
111
130
|
|
|
112
131
|
- name: Run argus-reviewer
|
|
113
132
|
id: run
|
|
133
|
+
if: inputs.run != 'false'
|
|
114
134
|
shell: bash
|
|
115
135
|
working-directory: ${{ inputs.working-directory }}
|
|
116
136
|
continue-on-error: true
|
|
117
137
|
env:
|
|
118
138
|
OPENROUTER_API_KEY: ${{ inputs.openrouter-api-key }}
|
|
139
|
+
GITHUB_TOKEN: ${{ github.token }}
|
|
119
140
|
ARGUS_DEBUG: '1'
|
|
120
141
|
ARGUS_DIFF_BASE: ${{ inputs.diff-base }}
|
|
121
142
|
ARGUS_BUDGET_USD: ${{ inputs.budget-usd }}
|
|
@@ -135,6 +156,7 @@ runs:
|
|
|
135
156
|
OPENROUTER_API_KEY: ${{ inputs.openrouter-api-key }}
|
|
136
157
|
VISION_E2E_WORKING_DIR: ${{ inputs.working-directory }}
|
|
137
158
|
ARGUS_REPORT_DIR: ${{ inputs.report-dir }}
|
|
159
|
+
ARGUS_RUN_DISABLED: ${{ inputs.run == 'false' && '1' || '' }}
|
|
138
160
|
with:
|
|
139
161
|
github-token: ${{ github.token }}
|
|
140
162
|
script: |
|
|
@@ -9,13 +9,23 @@ function formatUsd(n) {
|
|
|
9
9
|
return `$${(n || 0).toFixed(6)}`
|
|
10
10
|
}
|
|
11
11
|
|
|
12
|
+
/** Escape a report string for one markdown table cell. */
|
|
13
|
+
function cell(s) {
|
|
14
|
+
return String(s ?? '')
|
|
15
|
+
.replace(/\|/g, '\\|')
|
|
16
|
+
.replace(/[\r\n]+/g, ' ')
|
|
17
|
+
.slice(0, 200)
|
|
18
|
+
}
|
|
19
|
+
|
|
12
20
|
function renderMissingKeyBody() {
|
|
13
21
|
const lines = []
|
|
14
22
|
lines.push(SENTINEL)
|
|
15
23
|
lines.push('')
|
|
16
24
|
lines.push('## argus-reviewer ⚪ skipped')
|
|
17
25
|
lines.push('')
|
|
18
|
-
lines.push(
|
|
26
|
+
lines.push(
|
|
27
|
+
'`OPENROUTER_API_KEY` is not configured. Add it as a repository or workflow secret to run argus-reviewer.',
|
|
28
|
+
)
|
|
19
29
|
lines.push('')
|
|
20
30
|
lines.push('This status is intentionally neutral, not a failure.')
|
|
21
31
|
lines.push('')
|
|
@@ -28,14 +38,16 @@ function renderNoReportBody(reportDir, runUrl) {
|
|
|
28
38
|
lines.push('')
|
|
29
39
|
lines.push('## argus-reviewer ⚠️ no report')
|
|
30
40
|
lines.push('')
|
|
31
|
-
lines.push(
|
|
41
|
+
lines.push(
|
|
42
|
+
`The run step produced no \`run.json\` under \`${reportDir}\`. The commit status fails closed — check the action logs before merging.`,
|
|
43
|
+
)
|
|
32
44
|
lines.push('')
|
|
33
45
|
lines.push(`[View run](${runUrl})`)
|
|
34
46
|
lines.push('')
|
|
35
47
|
return lines.join('\n')
|
|
36
48
|
}
|
|
37
49
|
|
|
38
|
-
function renderBody(report, codeReview, runUrl, ok) {
|
|
50
|
+
function renderBody(report, codeReview, runUrl, ok, inlinePlan) {
|
|
39
51
|
if (!report) return renderMissingKeyBody()
|
|
40
52
|
|
|
41
53
|
const lines = []
|
|
@@ -68,7 +80,9 @@ function renderBody(report, codeReview, runUrl, ok) {
|
|
|
68
80
|
lines.push(`- \`${path.basename(t.file)}\` — ${t.name}`)
|
|
69
81
|
}
|
|
70
82
|
lines.push('')
|
|
71
|
-
lines.push(
|
|
83
|
+
lines.push(
|
|
84
|
+
`**Risk:** ${ok ? 'Low — UI regression tests and code review passed; no heals or failures.' : 'High — investigate failures before merge.'}`,
|
|
85
|
+
)
|
|
72
86
|
lines.push('')
|
|
73
87
|
if (Object.keys(trace).length > 0) {
|
|
74
88
|
lines.push('**Trace**')
|
|
@@ -87,7 +101,9 @@ function renderBody(report, codeReview, runUrl, ok) {
|
|
|
87
101
|
lines.push('| --- | --- | ---: | ---: | ---: | ---: |')
|
|
88
102
|
for (const t of report.tests) {
|
|
89
103
|
const result = t.ok ? '✅ pass' : '❌ fail'
|
|
90
|
-
lines.push(
|
|
104
|
+
lines.push(
|
|
105
|
+
`| ${t.name} | ${result} | ${t.visionCalls} | ${formatUsd(t.visionCostUsd)} | ${t.healEvents?.length ?? 0} | ${t.asserts?.length ?? 0} |`,
|
|
106
|
+
)
|
|
91
107
|
}
|
|
92
108
|
lines.push('')
|
|
93
109
|
lines.push('</details>')
|
|
@@ -173,14 +189,24 @@ function renderBody(report, codeReview, runUrl, ok) {
|
|
|
173
189
|
lines.push('')
|
|
174
190
|
lines.push('| Check | Status | Explanation |')
|
|
175
191
|
lines.push('| --- | --- | --- |')
|
|
176
|
-
lines.push(
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
lines.push(
|
|
192
|
+
lines.push(
|
|
193
|
+
`| Tests | ${report.ok ? '✅ Passed' : '❌ Failed'} | ${report.totals.passed}/${report.totals.tests} tests passed |`,
|
|
194
|
+
)
|
|
195
|
+
lines.push(
|
|
196
|
+
`| Budget | ${report.totals.budgetExceeded ? '⚠️ Warning' : '✅ Passed'} | ${formatUsd(report.totals.visionCostUsd)} spent${budgetCap > 0 ? ` of ${formatUsd(budgetCap)}` : ''} |`,
|
|
197
|
+
)
|
|
198
|
+
lines.push(
|
|
199
|
+
`| Heal events | ${healCount === 0 ? '✅ Passed' : '⚠️ Warning'} | ${healCount} heal event${healCount === 1 ? '' : 's'} |`,
|
|
200
|
+
)
|
|
201
|
+
lines.push(
|
|
202
|
+
`| Assertions | ${assertFails === 0 ? '✅ Passed' : '❌ Failed'} | ${assertFails === 0 ? assertCount : `${assertFails} failed`} assertion${assertCount === 1 ? '' : 's'} |`,
|
|
203
|
+
)
|
|
180
204
|
lines.push(`| OpenRouter key | ✅ Passed | \`OPENROUTER_API_KEY\` configured |`)
|
|
181
205
|
if (codeReview && !codeReview.skipped) {
|
|
182
206
|
const codeStatus = codeReview.ok ? '✅ Passed' : '❌ Failed'
|
|
183
|
-
lines.push(
|
|
207
|
+
lines.push(
|
|
208
|
+
`| Code review | ${codeStatus} | ${codeReview.findings.length} findings (${codeReview.model}) |`,
|
|
209
|
+
)
|
|
184
210
|
} else {
|
|
185
211
|
lines.push(`| Code review | ⚪ Skipped | ${codeReview?.summary ?? 'no report'} |`)
|
|
186
212
|
}
|
|
@@ -188,29 +214,7 @@ function renderBody(report, codeReview, runUrl, ok) {
|
|
|
188
214
|
lines.push('</details>')
|
|
189
215
|
lines.push('')
|
|
190
216
|
|
|
191
|
-
|
|
192
|
-
lines.push('<details>')
|
|
193
|
-
lines.push('<summary>🧠 Code review</summary>')
|
|
194
|
-
lines.push('')
|
|
195
|
-
lines.push(`**Verdict:** ${codeReview.verdict} · ${codeReview.model} · ${codeReview.tokens}tok ${formatUsd(codeReview.visionCostUsd)}`)
|
|
196
|
-
lines.push('')
|
|
197
|
-
lines.push(codeReview.summary)
|
|
198
|
-
lines.push('')
|
|
199
|
-
if (codeReview.findings.length > 0) {
|
|
200
|
-
const evidenceIcon = { exercised: '✅', corroborated: '🔴', not_exercised: '⚪', inconclusive: '❔' }
|
|
201
|
-
lines.push('| File | Severity | Evidence | Finding |')
|
|
202
|
-
lines.push('| --- | --- | --- | --- |')
|
|
203
|
-
for (const f of codeReview.findings) {
|
|
204
|
-
const ev = f.evidence
|
|
205
|
-
? `${evidenceIcon[f.evidence.status] ?? '❔'} ${f.evidence.detail}`
|
|
206
|
-
: '—'
|
|
207
|
-
lines.push(`| \`${f.file}\` | ${f.severity} | ${ev} | ${f.message} |`)
|
|
208
|
-
}
|
|
209
|
-
lines.push('')
|
|
210
|
-
}
|
|
211
|
-
lines.push('</details>')
|
|
212
|
-
lines.push('')
|
|
213
|
-
}
|
|
217
|
+
pushCodeReviewDetails(lines, codeReview, inlinePlan)
|
|
214
218
|
|
|
215
219
|
lines.push('<details>')
|
|
216
220
|
lines.push('<summary>✨ Actions</summary>')
|
|
@@ -228,50 +232,182 @@ function renderBody(report, codeReview, runUrl, ok) {
|
|
|
228
232
|
return lines.join('\n')
|
|
229
233
|
}
|
|
230
234
|
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
process.env.ARGUS_REPORT_DIR || 'argus-reviewer-report',
|
|
235
|
+
// The 🧠 Code review details block — shared by the full body and the
|
|
236
|
+
// review-only body (run lane disabled).
|
|
237
|
+
function pushCodeReviewDetails(lines, codeReview, inlinePlan) {
|
|
238
|
+
if (!codeReview || codeReview.skipped) return
|
|
239
|
+
lines.push('<details>')
|
|
240
|
+
lines.push('<summary>🧠 Code review</summary>')
|
|
241
|
+
lines.push('')
|
|
242
|
+
lines.push(
|
|
243
|
+
`**Verdict:** ${codeReview.verdict} · ${codeReview.model} · ${codeReview.tokens}tok ${formatUsd(codeReview.visionCostUsd)}`,
|
|
241
244
|
)
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
245
|
+
// U7 triage record — Jev annotate/route signals, never the gate.
|
|
246
|
+
if (codeReview.triage) {
|
|
247
|
+
const t = codeReview.triage
|
|
248
|
+
if (t.unadjudicated === true) {
|
|
249
|
+
lines.push(` · 🧭 triage unadjudicated — Jev unavailable`)
|
|
250
|
+
} else {
|
|
251
|
+
lines.push(
|
|
252
|
+
` · 🧭 triage: risk ${t.risk ?? '?'}/5` +
|
|
253
|
+
`${typeof t.needsDeepReview === 'number' ? ` · deep-review ${t.needsDeepReview.toFixed(2)}` : ''}` +
|
|
254
|
+
`${t.topRiskArea !== undefined ? ` · top area \`${cell(t.topRiskArea)}\`` : ''}` +
|
|
255
|
+
` (${cell(t.mode)})`,
|
|
256
|
+
)
|
|
252
257
|
}
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
codeReview
|
|
258
|
+
}
|
|
259
|
+
if (Array.isArray(codeReview.probes) && codeReview.probes.length > 0) {
|
|
260
|
+
const reproduced = codeReview.probes.filter((p) => p.outcome === 'reproduced').length
|
|
261
|
+
lines.push(
|
|
262
|
+
` · 🧪 ${codeReview.probes.length} probe${codeReview.probes.length === 1 ? '' : 's'} run, ${reproduced} reproduced`,
|
|
263
|
+
)
|
|
264
|
+
}
|
|
265
|
+
if (typeof codeReview.probeLaneSkipped === 'string') {
|
|
266
|
+
lines.push(` · 🧪 probe lane skipped — ${cell(codeReview.probeLaneSkipped)}`)
|
|
267
|
+
}
|
|
268
|
+
lines.push('')
|
|
269
|
+
lines.push(codeReview.summary)
|
|
270
|
+
lines.push('')
|
|
271
|
+
if (codeReview.findings.length > 0) {
|
|
272
|
+
const evidenceIcon = {
|
|
273
|
+
exercised: '✅',
|
|
274
|
+
corroborated: '🔴',
|
|
275
|
+
not_exercised: '⚪',
|
|
276
|
+
inconclusive: '❔',
|
|
277
|
+
reproduced: '🧪',
|
|
278
|
+
}
|
|
279
|
+
lines.push('| File | Severity | p | Category | Evidence | Finding |')
|
|
280
|
+
lines.push('| --- | --- | --- | --- | --- | --- |')
|
|
281
|
+
// Findings/evidence strings are model- and probe-emitted — sanitize
|
|
282
|
+
// for the markdown table and bound the section so an oversized report
|
|
283
|
+
// can't push the body past GitHub's 65536-char comment limit.
|
|
284
|
+
const MAX_FINDING_ROWS = 25
|
|
285
|
+
for (const f of codeReview.findings.slice(0, MAX_FINDING_ROWS)) {
|
|
286
|
+
const ev = f.evidence
|
|
287
|
+
? `${evidenceIcon[f.evidence.status] ?? '❔'} ${cell(f.evidence.detail)}`
|
|
288
|
+
: '—'
|
|
289
|
+
// U8 — Jev P(true positive); unadjudicated findings render '—'.
|
|
290
|
+
const p = typeof f.p === 'number' ? f.p.toFixed(2) : '—'
|
|
291
|
+
lines.push(
|
|
292
|
+
`| \`${cell(f.file)}\` | ${cell(f.severity)} | ${p} | ${cell(f.category ?? '—')} | ${ev} | ${cell(f.message)} |`,
|
|
293
|
+
)
|
|
294
|
+
}
|
|
295
|
+
if (codeReview.findings.length > MAX_FINDING_ROWS) {
|
|
296
|
+
lines.push(
|
|
297
|
+
`| … | — | — | — | — | ${codeReview.findings.length - MAX_FINDING_ROWS} more findings in \`code-review.json\` |`,
|
|
298
|
+
)
|
|
299
|
+
}
|
|
300
|
+
lines.push('')
|
|
301
|
+
// Inline-comment cap note — dedup'd fresh findings the
|
|
302
|
+
// review.maxComments budget didn't post (TCA max_comments).
|
|
303
|
+
if (inlinePlan !== undefined && inlinePlan.dropped > 0) {
|
|
304
|
+
lines.push(
|
|
305
|
+
`*+${inlinePlan.dropped} inline-eligible finding(s) not posted — \`review.maxComments\` cap ${inlinePlan.cap}.*`,
|
|
306
|
+
)
|
|
307
|
+
lines.push('')
|
|
308
|
+
}
|
|
309
|
+
// Secrets-lane audit line — adjudicated/suppressed counts, never literals.
|
|
310
|
+
if (codeReview.secretsScan) {
|
|
311
|
+
if (typeof codeReview.secretsScan.skipped === 'string') {
|
|
312
|
+
lines.push(`*🔐 secrets scan skipped — ${cell(codeReview.secretsScan.skipped)}*`)
|
|
313
|
+
} else if (Array.isArray(codeReview.secretsScan.records)) {
|
|
314
|
+
const suppressed = codeReview.secretsScan.records.filter((r) => r.suppressed).length
|
|
315
|
+
const unadj = codeReview.secretsScan.records.filter((r) => !r.adjudicated).length
|
|
316
|
+
lines.push(
|
|
317
|
+
`*🔐 secrets scan: ${codeReview.secretsScan.records.length} candidate(s)` +
|
|
318
|
+
`${suppressed > 0 ? `, ${suppressed} adjudicated-suppressed` : ''}` +
|
|
319
|
+
`${unadj > 0 ? `, ${unadj} unadjudicated` : ''}` +
|
|
320
|
+
`${codeReview.secretsScan.overflow > 0 ? `, +${codeReview.secretsScan.overflow} over cap` : ''}.*`,
|
|
321
|
+
)
|
|
322
|
+
}
|
|
323
|
+
lines.push('')
|
|
258
324
|
}
|
|
259
325
|
}
|
|
326
|
+
// U8 adjudication audit — outside the findings guard so suppressed-
|
|
327
|
+
// only reviews still show what Jev removed. p values live on the
|
|
328
|
+
// findings table and in code-review.json records.
|
|
329
|
+
if (codeReview.findingAdjudication && Array.isArray(codeReview.findingAdjudication.records)) {
|
|
330
|
+
const fa = codeReview.findingAdjudication
|
|
331
|
+
if (fa.unadjudicated === true) {
|
|
332
|
+
lines.push('*🧮 adjudication: unadjudicated — Jev unavailable, nothing suppressed.*')
|
|
333
|
+
} else {
|
|
334
|
+
const suppressed = fa.records.filter((r) => r.suppressed).length
|
|
335
|
+
const unadj = fa.records.filter((r) => !r.adjudicated).length
|
|
336
|
+
lines.push(
|
|
337
|
+
`*🧮 adjudication: ${fa.records.length} finding(s) scored` +
|
|
338
|
+
`${suppressed > 0 ? `, ${suppressed} suppressed (nit/q)` : ''}` +
|
|
339
|
+
`${unadj > 0 ? `, ${unadj} unadjudicated` : ''}` +
|
|
340
|
+
`${fa.overflow > 0 ? `, +${fa.overflow} over cap` : ''}.*`,
|
|
341
|
+
)
|
|
342
|
+
}
|
|
343
|
+
lines.push('')
|
|
344
|
+
}
|
|
345
|
+
lines.push('</details>')
|
|
346
|
+
lines.push('')
|
|
347
|
+
}
|
|
260
348
|
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
const
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
349
|
+
// Sticky body for `run: 'false'` consumers — no run.json exists by
|
|
350
|
+
// design, so the body and conclusion reflect code-review alone.
|
|
351
|
+
function renderReviewOnlyBody(codeReview, runUrl, ok, inlinePlan) {
|
|
352
|
+
const lines = []
|
|
353
|
+
lines.push(SENTINEL)
|
|
354
|
+
lines.push('')
|
|
355
|
+
lines.push(`## argus-reviewer ${ok ? '✅ PASS' : '❌ FAIL'}`)
|
|
356
|
+
lines.push('')
|
|
357
|
+
if (!codeReview) {
|
|
358
|
+
lines.push(
|
|
359
|
+
'**Summary:** code-review only (run lane disabled) — no `code-review.json` found. The review step crashed or produced no report; the commit status fails closed — check the action logs before merging.',
|
|
360
|
+
)
|
|
361
|
+
} else if (codeReview.skipped) {
|
|
362
|
+
lines.push(
|
|
363
|
+
`**Summary:** code-review only (run lane disabled) — review skipped: ${cell(codeReview.summary)}`,
|
|
364
|
+
)
|
|
365
|
+
} else {
|
|
366
|
+
lines.push(
|
|
367
|
+
`**Summary:** code review only (run lane disabled) · verdict **${codeReview.verdict}** · ` +
|
|
368
|
+
`${codeReview.findings.length} finding(s) · ${codeReview.model} · ` +
|
|
369
|
+
`${codeReview.tokens}tok ${formatUsd(codeReview.visionCostUsd)}`,
|
|
370
|
+
)
|
|
371
|
+
}
|
|
372
|
+
lines.push('')
|
|
373
|
+
lines.push('<details>')
|
|
374
|
+
lines.push('<summary>🚥 Pre-merge checks</summary>')
|
|
375
|
+
lines.push('')
|
|
376
|
+
lines.push('| Check | Status | Explanation |')
|
|
377
|
+
lines.push('| --- | --- | --- |')
|
|
378
|
+
lines.push('| OpenRouter key | ✅ Passed | `OPENROUTER_API_KEY` configured |')
|
|
379
|
+
if (codeReview && !codeReview.skipped) {
|
|
380
|
+
const codeStatus = codeReview.ok ? '✅ Passed' : '❌ Failed'
|
|
381
|
+
lines.push(
|
|
382
|
+
`| Code review | ${codeStatus} | ${codeReview.findings.length} findings (${codeReview.model}) |`,
|
|
383
|
+
)
|
|
384
|
+
} else {
|
|
385
|
+
lines.push(
|
|
386
|
+
`| Code review | ${codeReview ? '⚪ Skipped' : '❌ Failed'} | ${cell(codeReview?.summary ?? 'no report')} |`,
|
|
387
|
+
)
|
|
388
|
+
}
|
|
389
|
+
lines.push('')
|
|
390
|
+
lines.push('</details>')
|
|
391
|
+
lines.push('')
|
|
392
|
+
pushCodeReviewDetails(lines, codeReview, inlinePlan)
|
|
393
|
+
if (runUrl) lines.push(`[View run](${runUrl})`)
|
|
394
|
+
lines.push('')
|
|
395
|
+
lines.push('<details>')
|
|
396
|
+
lines.push('<summary>✨ Actions</summary>')
|
|
397
|
+
lines.push('')
|
|
398
|
+
lines.push('- [ ] Re-run argus-reviewer')
|
|
399
|
+
lines.push('')
|
|
400
|
+
lines.push('</details>')
|
|
401
|
+
lines.push('')
|
|
402
|
+
lines.push('---')
|
|
403
|
+
lines.push('')
|
|
404
|
+
lines.push('<sub>`argus-reviewer` — self-hosted, BYOK OpenRouter UI regression.</sub>')
|
|
405
|
+
lines.push('')
|
|
406
|
+
return lines.join('\n')
|
|
407
|
+
}
|
|
272
408
|
|
|
273
|
-
async function
|
|
274
|
-
if (!pr || !codeReview || codeReview.skipped || !codeReview.findings) return
|
|
409
|
+
async function planInlineComments(pr, codeReview) {
|
|
410
|
+
if (!pr || !codeReview || codeReview.skipped || !codeReview.findings) return undefined
|
|
275
411
|
// Must match the severity vocabulary emitted by the code-review schema
|
|
276
412
|
// (src/cli.ts): bug/risk are inline-worthy; nit/q stay in the sticky body.
|
|
277
413
|
const inlineSeverities = ['bug', 'risk']
|
|
@@ -281,14 +417,25 @@ async function postInlineComments(pr, codeReview) {
|
|
|
281
417
|
path: f.file,
|
|
282
418
|
line: f.line,
|
|
283
419
|
side: 'RIGHT',
|
|
284
|
-
body: `**argus-reviewer ${f.severity}:** ${f.message}${
|
|
420
|
+
body: `**argus-reviewer ${f.severity}:** ${f.message}${
|
|
421
|
+
f.category ? ` \`${f.category}\`` : ''
|
|
422
|
+
}${
|
|
423
|
+
f.evidence && f.evidence.status === 'reproduced'
|
|
424
|
+
? '\n\n*🧪 Reproduced by an Argus probe — fails on this PR head, clean on base. See workflow artifacts.*'
|
|
425
|
+
: f.evidence && f.evidence.status !== 'exercised'
|
|
426
|
+
? `\n\n*CI evidence: ${f.evidence.detail}*`
|
|
427
|
+
: ''
|
|
428
|
+
}`,
|
|
285
429
|
}))
|
|
286
|
-
if (comments.length === 0) return
|
|
430
|
+
if (comments.length === 0) return { capped: [], dropped: 0, cap: 0 }
|
|
287
431
|
|
|
288
432
|
// Re-runs on the same SHA must not duplicate inline comments — the sticky
|
|
289
433
|
// body is upserted but review comments are not. Paginate fully (100/page)
|
|
290
434
|
// and scope dedup to the current head: comments on older commits must not
|
|
291
|
-
// suppress findings that still apply to this head.
|
|
435
|
+
// suppress findings that still apply to this head. The key is the body's
|
|
436
|
+
// first line (the finding itself) — trailing evidence notes like the
|
|
437
|
+
// `reproduced` upgrade change the body but must not re-post a duplicate.
|
|
438
|
+
const dedupKey = (path, line, body) => `${path}:${line}:${body.split('\n')[0]}`
|
|
292
439
|
const posted = new Set()
|
|
293
440
|
let page = 1
|
|
294
441
|
for (;;) {
|
|
@@ -301,15 +448,24 @@ async function postInlineComments(pr, codeReview) {
|
|
|
301
448
|
})
|
|
302
449
|
for (const c of existing) {
|
|
303
450
|
if (c.body && c.body.startsWith('**argus-reviewer') && c.commit_id === pr.head.sha) {
|
|
304
|
-
posted.add(
|
|
451
|
+
posted.add(dedupKey(c.path, c.line, c.body))
|
|
305
452
|
}
|
|
306
453
|
}
|
|
307
454
|
if (existing.length < 100) break
|
|
308
455
|
page += 1
|
|
309
456
|
}
|
|
310
|
-
const fresh = comments.filter((c) => !posted.has(
|
|
311
|
-
|
|
457
|
+
const fresh = comments.filter((c) => !posted.has(dedupKey(c.path, c.line, c.body)))
|
|
458
|
+
|
|
459
|
+
// review.maxComments caps inline noise (TCA max_comments) — applied
|
|
460
|
+
// after dedup so already-posted comments don't eat the budget. A cap of
|
|
461
|
+
// 0 disables inline posting entirely (no empty review).
|
|
462
|
+
const cap = typeof codeReview.maxComments === 'number' ? codeReview.maxComments : 20
|
|
463
|
+
const capped = fresh.slice(0, cap)
|
|
464
|
+
return { capped, dropped: fresh.length - capped.length, cap }
|
|
465
|
+
}
|
|
312
466
|
|
|
467
|
+
async function postInlineComments(pr, plan) {
|
|
468
|
+
if (plan === undefined || plan.capped.length === 0) return
|
|
313
469
|
// One batched review instead of N createReviewComment calls — avoids
|
|
314
470
|
// secondary rate limits on large findings sets.
|
|
315
471
|
try {
|
|
@@ -319,13 +475,65 @@ async function postInlineComments(pr, codeReview) {
|
|
|
319
475
|
pull_number: pr.number,
|
|
320
476
|
commit_id: pr.head.sha,
|
|
321
477
|
event: 'COMMENT',
|
|
322
|
-
comments:
|
|
478
|
+
comments: plan.capped,
|
|
323
479
|
})
|
|
324
480
|
} catch (e) {
|
|
325
481
|
core.warning(`inline review failed: ${e.message}`)
|
|
326
482
|
}
|
|
327
483
|
}
|
|
328
484
|
|
|
485
|
+
async function main() {
|
|
486
|
+
const pr = context.payload && context.payload.pull_request
|
|
487
|
+
const owner = context.repo.owner
|
|
488
|
+
const repo = context.repo.repo
|
|
489
|
+
const hasKey = !!process.env.OPENROUTER_API_KEY
|
|
490
|
+
const workDir = process.env.VISION_E2E_WORKING_DIR || ''
|
|
491
|
+
const reportDir = path.resolve(
|
|
492
|
+
process.env.GITHUB_WORKSPACE,
|
|
493
|
+
workDir,
|
|
494
|
+
process.env.ARGUS_REPORT_DIR || 'argus-reviewer-report',
|
|
495
|
+
)
|
|
496
|
+
const runUrl = `${process.env.GITHUB_SERVER_URL}/${owner}/${repo}/actions/runs/${process.env.GITHUB_RUN_ID}`
|
|
497
|
+
|
|
498
|
+
let report
|
|
499
|
+
let codeReview
|
|
500
|
+
if (hasKey) {
|
|
501
|
+
try {
|
|
502
|
+
const raw = fs.readFileSync(path.join(reportDir, 'run.json'), 'utf8')
|
|
503
|
+
report = JSON.parse(raw)
|
|
504
|
+
} catch {
|
|
505
|
+
report = undefined
|
|
506
|
+
}
|
|
507
|
+
try {
|
|
508
|
+
const raw = fs.readFileSync(path.join(reportDir, 'code-review.json'), 'utf8')
|
|
509
|
+
codeReview = JSON.parse(raw)
|
|
510
|
+
} catch {
|
|
511
|
+
codeReview = undefined
|
|
512
|
+
}
|
|
513
|
+
}
|
|
514
|
+
|
|
515
|
+
// Missing code-review.json after a continue-on-error step means the review
|
|
516
|
+
// crashed, not that it skipped — an intentional skip writes ok+skipped.
|
|
517
|
+
// Fail closed rather than reporting it as a clean skip.
|
|
518
|
+
const codeReviewOk = codeReview != null && codeReview.ok === true
|
|
519
|
+
// run: 'false' consumers have no run.json by design — the conclusion then
|
|
520
|
+
// reflects the code-review verdict alone.
|
|
521
|
+
const runDisabled = process.env.ARGUS_RUN_DISABLED === '1'
|
|
522
|
+
const ok = (runDisabled || report?.ok === true) && codeReviewOk
|
|
523
|
+
const conclusion = !hasKey ? 'neutral' : ok ? 'success' : 'failure'
|
|
524
|
+
const inlinePlan = hasKey ? await planInlineComments(pr, codeReview) : undefined
|
|
525
|
+
const body = !hasKey
|
|
526
|
+
? renderMissingKeyBody()
|
|
527
|
+
: runDisabled
|
|
528
|
+
? renderReviewOnlyBody(codeReview, runUrl, ok, inlinePlan)
|
|
529
|
+
: report === undefined
|
|
530
|
+
? renderNoReportBody(reportDir, runUrl)
|
|
531
|
+
: renderBody(report, codeReview, runUrl, ok, inlinePlan)
|
|
532
|
+
|
|
533
|
+
// Eligibility + dedup for inline comments, computed before the sticky
|
|
534
|
+
// body renders so the "+N not posted" note counts the *fresh* set — the
|
|
535
|
+
// cap is applied to fresh, not to raw findings (already-posted comments
|
|
536
|
+
// must not inflate the dropped count).
|
|
329
537
|
if (pr) {
|
|
330
538
|
const { data: comments } = await github.rest.issues.listComments({
|
|
331
539
|
owner,
|
|
@@ -349,7 +557,7 @@ async function postInlineComments(pr, codeReview) {
|
|
|
349
557
|
body,
|
|
350
558
|
})
|
|
351
559
|
}
|
|
352
|
-
await postInlineComments(pr,
|
|
560
|
+
await postInlineComments(pr, inlinePlan)
|
|
353
561
|
}
|
|
354
562
|
|
|
355
563
|
const sha = pr ? pr.head.sha : context.sha
|
package/dist/api.d.ts
CHANGED
|
@@ -7,6 +7,7 @@ import { Config, defineConfig } from './config.js';
|
|
|
7
7
|
import { ErrorRecord } from './journal/schema.js';
|
|
8
8
|
import { Logger } from './log.js';
|
|
9
9
|
export { defineConfig };
|
|
10
|
+
export { DecisionClient, DecisionError, JEV_DEFAULT_MODEL, type DecisionAnswer, type DecisionClientOptions, type DecisionErrorKind, type DecisionQuestion, } from './vision/decisions.js';
|
|
10
11
|
/**
|
|
11
12
|
* Test-facing API (R13). Test files are plain TypeScript using a `td` object:
|
|
12
13
|
*
|
package/dist/api.js
CHANGED
|
@@ -4,6 +4,7 @@ import { loadFlow, saveFlow } from './cache/store.js';
|
|
|
4
4
|
import { Ledger } from './vision/ledger.js';
|
|
5
5
|
import { defineConfig } from './config.js';
|
|
6
6
|
export { defineConfig };
|
|
7
|
+
export { DecisionClient, DecisionError, JEV_DEFAULT_MODEL, } from './vision/decisions.js';
|
|
7
8
|
const KEY_ALIASES = {
|
|
8
9
|
tab: 'Tab',
|
|
9
10
|
enter: 'Enter',
|