argus-reviewer-e2e 0.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (67) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +80 -0
  3. package/action/action.yml +147 -0
  4. package/action/sticky-comment.mjs +376 -0
  5. package/dist/api.d.ts +121 -0
  6. package/dist/api.js +256 -0
  7. package/dist/cache/fingerprint.d.ts +59 -0
  8. package/dist/cache/fingerprint.js +67 -0
  9. package/dist/cache/store.d.ts +17 -0
  10. package/dist/cache/store.js +41 -0
  11. package/dist/cli.d.ts +25 -0
  12. package/dist/cli.js +1355 -0
  13. package/dist/config.d.ts +130 -0
  14. package/dist/config.js +163 -0
  15. package/dist/debug.d.ts +1 -0
  16. package/dist/debug.js +30 -0
  17. package/dist/detect.d.ts +50 -0
  18. package/dist/detect.js +105 -0
  19. package/dist/driver/browser.d.ts +64 -0
  20. package/dist/driver/browser.js +200 -0
  21. package/dist/driver/target.d.ts +23 -0
  22. package/dist/driver/target.js +97 -0
  23. package/dist/engine/actions.d.ts +26 -0
  24. package/dist/engine/actions.js +47 -0
  25. package/dist/engine/loop.d.ts +118 -0
  26. package/dist/engine/loop.js +649 -0
  27. package/dist/engine/prompts.d.ts +22 -0
  28. package/dist/engine/prompts.js +112 -0
  29. package/dist/evidence/ci.d.ts +18 -0
  30. package/dist/evidence/ci.js +61 -0
  31. package/dist/evidence/link.d.ts +35 -0
  32. package/dist/evidence/link.js +90 -0
  33. package/dist/executor/a0.d.ts +29 -0
  34. package/dist/executor/a0.js +38 -0
  35. package/dist/fsutil.d.ts +5 -0
  36. package/dist/fsutil.js +12 -0
  37. package/dist/index/context.d.ts +17 -0
  38. package/dist/index/context.js +88 -0
  39. package/dist/index/diff.d.ts +1 -0
  40. package/dist/index/diff.js +33 -0
  41. package/dist/index/invalidate.d.ts +28 -0
  42. package/dist/index/invalidate.js +56 -0
  43. package/dist/index/scan.d.ts +26 -0
  44. package/dist/index/scan.js +209 -0
  45. package/dist/journal/build.d.ts +15 -0
  46. package/dist/journal/build.js +47 -0
  47. package/dist/journal/schema.d.ts +59 -0
  48. package/dist/journal/schema.js +6 -0
  49. package/dist/journal/store.d.ts +10 -0
  50. package/dist/journal/store.js +26 -0
  51. package/dist/live.d.ts +2 -0
  52. package/dist/live.js +46 -0
  53. package/dist/log.d.ts +17 -0
  54. package/dist/log.js +24 -0
  55. package/dist/report/comment.d.ts +13 -0
  56. package/dist/report/comment.js +135 -0
  57. package/dist/report/junit.d.ts +10 -0
  58. package/dist/report/junit.js +46 -0
  59. package/dist/report/run.d.ts +51 -0
  60. package/dist/report/run.js +36 -0
  61. package/dist/vision/cost.d.ts +36 -0
  62. package/dist/vision/cost.js +16 -0
  63. package/dist/vision/ledger.d.ts +29 -0
  64. package/dist/vision/ledger.js +65 -0
  65. package/dist/vision/openrouter.d.ts +70 -0
  66. package/dist/vision/openrouter.js +134 -0
  67. package/package.json +65 -0
package/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Luke Kimball
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
package/README.md ADDED
@@ -0,0 +1,80 @@
1
+ # argus-reviewer
2
+
3
+ <p align="center">
4
+ <img src="docs/assets/social.png" alt="Argus — vision-model E2E testing" width="640" />
5
+ </p>
6
+
7
+ Open-source, self-hosted vision-model E2E testing — the hundred-eyed watcher for your UI. Bring your own `OPENROUTER_API_KEY`: record a flow once, fingerprint-cache every step, replay near-free, heal on UI drift, and get results as a check + comment on the GitHub PR.
8
+
9
+ - **Vision-first**: a model looks at a screenshot and decides where to click — no selectors to write or maintain.
10
+ - **Cache-first**: replay costs zero vision calls on an unchanged UI; heals re-spend only on drift and show up as reviewable cache diffs.
11
+ - **Cost-explicit**: every call is metered from OpenRouter's per-call cost and rolled into a per-run dollar figure on the PR.
12
+ - **Grounding specialist**: a `grounding_model` (e.g. a ui-tars-class model) can drive element location with its native coordinate output, verified against the DOM before any click executes.
13
+
14
+ ```bash
15
+ npm i -D argus-reviewer-e2e # or github:duketopceo/argus-reviewer
16
+ npx argus-reviewer record "log in and open settings" --url https://localhost:3000
17
+ npx argus-reviewer run # replays + asserts, zero-cost on cache hit
18
+ ```
19
+
20
+ <p align="center">
21
+ <img src="docs/assets/demo.gif" alt="argus-reviewer run — live vision call, PASS, $0.0005 spend" width="900" />
22
+ </p>
23
+
24
+ *Real `run` output: one vision assert, `PASS`, and the exact dollar figure on the run report.*
25
+
26
+ Configuration lives in `argus-reviewer.config.ts` (a legacy `vision-e2e.config.*` is still accepted) — see `src/config.ts` for the full shape: `model`, `grounding_model`, `escalation_model`, `provider` routing rules, `budgetUsd`, `target`, `pageSetup`, `secrets`.
27
+
28
+ ### OpenRouter cost attribution
29
+
30
+ Add an `openrouter` block to tag every request. `trace` is sent in the request body and is the right hook for cost allocation by repo/PR/run. `headers` are sent verbatim with every OpenRouter request (useful for `HTTP-Referer` or `X-Title`).
31
+
32
+ ```ts
33
+ export default {
34
+ openrouter: {
35
+ trace: { repo: 'duketopceo/myapp', pr: '42', run: 'argus-reviewer' },
36
+ headers: { 'HTTP-Referer': 'https://github.com/duketopceo/myapp' },
37
+ },
38
+ }
39
+ ```
40
+
41
+ The GitHub Action automatically sets `ARGUS_REVIEWER_TRACE` with the repository, PR number, commit, and run id, so every PR review is attributed in OpenRouter without extra config. You can also set `ARGUS_REVIEWER_TRACE` yourself (JSON object) to add more fields.
42
+
43
+ Status: early development. See `action/` for the composite GitHub Action and `runner/` for self-hosted runner registration.
44
+
45
+ ## File structure
46
+
47
+ ```text
48
+ argus-reviewer/
49
+ ├── action/ # GitHub Actions composite action + sticky PR comment
50
+ │ ├── action.yml
51
+ │ └── sticky-comment.mjs
52
+ ├── runner/ # Self-hosted runner registration docs + script
53
+ │ ├── README.md
54
+ │ └── register-runner.sh
55
+ ├── src/
56
+ │ ├── api.ts # Test-facing `test`/`td` API + generated test file renderer
57
+ │ ├── cli.ts # `record`, `run`, and `cache` commands
58
+ │ ├── config.ts # `argus-reviewer.config.*` loader (legacy `vision-e2e.config.*` accepted)
59
+ │ ├── driver/
60
+ │ │ ├── browser.ts # Playwright browser launch (chromium/firefox/webkit) + observation capture
61
+ │ │ └── target.ts # Optional local dev-server target process
62
+ │ ├── engine/
63
+ │ │ ├── actions.ts # Low-level page actions (click, type, scroll, …)
64
+ │ │ ├── loop.ts # Vision model record/replay + healing loop
65
+ │ │ └── prompts.ts # OpenRouter action/assertion prompts + JSON schemas
66
+ │ ├── cache/
67
+ │ │ ├── fingerprint.ts # Per-step screenshot/a11y fingerprint + resolve
68
+ │ │ └── store.ts # Flow cache read/write
69
+ │ ├── report/
70
+ │ │ ├── comment.ts # Markdown PR comment + commit-status rendering
71
+ │ │ ├── junit.ts # JUnit XML output
72
+ │ │ └── run.ts # JSON run report consumed by the action
73
+ │ └── vision/
74
+ │ ├── cost.ts # OpenRouter cost parsing per call
75
+ │ ├── ledger.ts # Per-run USD budget tracking
76
+ │ └── openrouter.ts # OpenRouter chat-completion client + schema parsing
77
+ └── tests/ # Unit tests + small Playwright fixture page
78
+ ```
79
+
80
+ License: MIT.
@@ -0,0 +1,147 @@
1
+ name: argus-reviewer
2
+ description: Run the argus-reviewer harness and post a sticky PR comment with cost and evidence.
3
+ author: duketopceo
4
+ inputs:
5
+ openrouter-api-key:
6
+ description: OpenRouter API key (BYOK). All vision calls are billed through this key.
7
+ required: true
8
+ cli:
9
+ description: Command used to run argus-reviewer (defaults to `npx --no-install argus-reviewer`). For dogfooding this repo, use `node dist/cli.js` after `npm run build`.
10
+ default: npx --no-install argus-reviewer
11
+ required: false
12
+ config:
13
+ description: Path to the argus-reviewer config file
14
+ default: argus-reviewer.config.ts
15
+ required: false
16
+ working-directory:
17
+ description: Working directory for the consumer's project
18
+ required: false
19
+ budget-usd:
20
+ description: Optional per-run budget cap override in USD
21
+ required: false
22
+ index:
23
+ description: Run `argus-reviewer index` before the test run to enable diff-aware cache invalidation
24
+ default: 'true'
25
+ required: false
26
+ diff-base:
27
+ description: Base ref for diff invalidation (e.g. origin/main); empty = working tree
28
+ default: ''
29
+ required: false
30
+ node-version:
31
+ description: Node.js version for the harness (Node 20 is deprecated on GH runners)
32
+ default: '22'
33
+ required: false
34
+ browser:
35
+ description: Playwright browser engine to install and run (chromium, firefox, or webkit). Should match config.browser.
36
+ default: 'chromium'
37
+ required: false
38
+ cache-dependency-path:
39
+ description: npm lockfile path for setup-node cache (defaults to <working-directory>/package-lock.json)
40
+ required: false
41
+ report-dir:
42
+ description: Report output dir, passed to both run and code-review so the post step can find run.json/code-review.json regardless of config reportDir
43
+ default: argus-reviewer-report
44
+ required: false
45
+ outputs:
46
+ conclusion:
47
+ description: 'Run conclusion: success, failure, or neutral'
48
+ value: ${{ steps.post.outputs.conclusion }}
49
+ runs:
50
+ using: composite
51
+ steps:
52
+ - uses: actions/setup-node@v4
53
+ with:
54
+ node-version: ${{ inputs.node-version }}
55
+ cache: npm
56
+ cache-dependency-path: ${{ inputs.cache-dependency-path || format('{0}/package-lock.json', inputs.working-directory || '.') }}
57
+
58
+ - name: Install consumer dependencies
59
+ shell: bash
60
+ working-directory: ${{ inputs.working-directory }}
61
+ run: npm ci
62
+
63
+ - name: Install Playwright browser
64
+ shell: bash
65
+ working-directory: ${{ inputs.working-directory }}
66
+ env:
67
+ ARGUS_BROWSER: ${{ inputs.browser }}
68
+ run: |
69
+ case "$ARGUS_BROWSER" in
70
+ chromium|firefox|webkit) ;;
71
+ *) echo "invalid browser input: '$ARGUS_BROWSER' (expected chromium, firefox, or webkit)"; exit 1 ;;
72
+ esac
73
+ npx playwright install --with-deps "$ARGUS_BROWSER" || npx playwright install "$ARGUS_BROWSER"
74
+
75
+ - name: Stage config
76
+ shell: bash
77
+ working-directory: ${{ inputs.working-directory }}
78
+ run: |
79
+ if [ "${{ inputs.config }}" != "argus-reviewer.config.ts" ]; then
80
+ ext="${{ inputs.config }}"
81
+ ext="${ext##*.}"
82
+ case "$ext" in
83
+ ts|json) cp "${{ inputs.config }}" "argus-reviewer.config.$ext" ;;
84
+ *) echo "unsupported config extension: $ext" >&2; exit 1 ;;
85
+ esac
86
+ fi
87
+
88
+ - name: Build repo index
89
+ if: inputs.index == 'true'
90
+ shell: bash
91
+ working-directory: ${{ inputs.working-directory }}
92
+ run: ${{ inputs.cli }} index
93
+
94
+ - name: Run argus-reviewer code review
95
+ id: code-review
96
+ shell: bash
97
+ working-directory: ${{ inputs.working-directory }}
98
+ continue-on-error: true
99
+ env:
100
+ OPENROUTER_API_KEY: ${{ inputs.openrouter-api-key }}
101
+ GITHUB_TOKEN: ${{ github.token }}
102
+ ARGUS_DEBUG: '1'
103
+ ARGUS_REVIEWER_TRACE: >-
104
+ {"repo":"${{ github.repository }}",
105
+ "pr":"${{ github.event.pull_request.number }}",
106
+ "commit":"${{ github.sha }}",
107
+ "run_id":"${{ github.run_id }}",
108
+ "run_attempt":"${{ github.run_attempt }}",
109
+ "workflow":"${{ github.workflow }}"}
110
+ run: ${{ inputs.cli }} code-review --report-dir "${{ inputs.report-dir }}"
111
+
112
+ - name: Run argus-reviewer
113
+ id: run
114
+ shell: bash
115
+ working-directory: ${{ inputs.working-directory }}
116
+ continue-on-error: true
117
+ env:
118
+ OPENROUTER_API_KEY: ${{ inputs.openrouter-api-key }}
119
+ ARGUS_DEBUG: '1'
120
+ ARGUS_DIFF_BASE: ${{ inputs.diff-base }}
121
+ ARGUS_BUDGET_USD: ${{ inputs.budget-usd }}
122
+ ARGUS_REVIEWER_TRACE: >-
123
+ {"repo":"${{ github.repository }}",
124
+ "pr":"${{ github.event.pull_request.number }}",
125
+ "commit":"${{ github.sha }}",
126
+ "run_id":"${{ github.run_id }}",
127
+ "run_attempt":"${{ github.run_attempt }}",
128
+ "workflow":"${{ github.workflow }}"}
129
+ run: ${{ inputs.cli }} run --report-dir "${{ inputs.report-dir }}"
130
+
131
+ - name: Post or update sticky PR comment
132
+ id: post
133
+ uses: actions/github-script@v7
134
+ env:
135
+ OPENROUTER_API_KEY: ${{ inputs.openrouter-api-key }}
136
+ VISION_E2E_WORKING_DIR: ${{ inputs.working-directory }}
137
+ ARGUS_REPORT_DIR: ${{ inputs.report-dir }}
138
+ with:
139
+ github-token: ${{ github.token }}
140
+ script: |
141
+ const fs = require('fs');
142
+ const path = require('path');
143
+ const actionPath = process.env.GITHUB_ACTION_PATH;
144
+ const file = path.join(actionPath, 'sticky-comment.mjs');
145
+ const code = fs.readFileSync(file, 'utf8');
146
+ const runFn = new Function('github', 'context', 'core', 'require', `return (async () => {\n${code}\n})()`);
147
+ return await runFn(github, context, core, require);
@@ -0,0 +1,376 @@
1
+ /* global github, context, core, require, process */
2
+
3
+ const fs = require('fs')
4
+ const path = require('path')
5
+
6
+ const SENTINEL = '<!-- argus-reviewer -->'
7
+
8
+ function formatUsd(n) {
9
+ return `$${(n || 0).toFixed(6)}`
10
+ }
11
+
12
+ function renderMissingKeyBody() {
13
+ const lines = []
14
+ lines.push(SENTINEL)
15
+ lines.push('')
16
+ lines.push('## argus-reviewer ⚪ skipped')
17
+ lines.push('')
18
+ lines.push('`OPENROUTER_API_KEY` is not configured. Add it as a repository or workflow secret to run argus-reviewer.')
19
+ lines.push('')
20
+ lines.push('This status is intentionally neutral, not a failure.')
21
+ lines.push('')
22
+ return lines.join('\n')
23
+ }
24
+
25
+ function renderNoReportBody(reportDir, runUrl) {
26
+ const lines = []
27
+ lines.push(SENTINEL)
28
+ lines.push('')
29
+ lines.push('## argus-reviewer ⚠️ no report')
30
+ lines.push('')
31
+ lines.push(`The run step produced no \`run.json\` under \`${reportDir}\`. The commit status fails closed — check the action logs before merging.`)
32
+ lines.push('')
33
+ lines.push(`[View run](${runUrl})`)
34
+ lines.push('')
35
+ return lines.join('\n')
36
+ }
37
+
38
+ function renderBody(report, codeReview, runUrl, ok) {
39
+ if (!report) return renderMissingKeyBody()
40
+
41
+ const lines = []
42
+ const budgetCap = report.config?.budgetUsd ?? 0
43
+ const healCount = report.tests.reduce((n, t) => n + (t.healEvents?.length ?? 0), 0)
44
+ const assertCount = report.tests.reduce((n, t) => n + (t.asserts?.length ?? 0), 0)
45
+ const assertFails = report.tests.reduce(
46
+ (n, t) => n + (t.asserts?.filter((a) => a.verdict === 'fail').length ?? 0),
47
+ 0,
48
+ )
49
+ const trace = report.trace ?? {}
50
+
51
+ lines.push(SENTINEL)
52
+ lines.push('')
53
+ lines.push(`## argus-reviewer ${ok ? '✅ PASS' : '❌ FAIL'}`)
54
+ lines.push('')
55
+ lines.push(
56
+ `**Summary:** ${report.totals.passed}/${report.totals.tests} passed · ` +
57
+ `${report.totals.visionCalls} vision calls · ` +
58
+ `${formatUsd(report.totals.visionCostUsd)} spend · ` +
59
+ `${report.totals.sandboxSeconds.toFixed(1)}s sandbox`,
60
+ )
61
+ lines.push('')
62
+
63
+ lines.push('<details>')
64
+ lines.push('<summary>📝 Summary</summary>')
65
+ lines.push('')
66
+ lines.push('**What ran**')
67
+ for (const t of report.tests) {
68
+ lines.push(`- \`${path.basename(t.file)}\` — ${t.name}`)
69
+ }
70
+ lines.push('')
71
+ lines.push(`**Risk:** ${ok ? 'Low — UI regression tests and code review passed; no heals or failures.' : 'High — investigate failures before merge.'}`)
72
+ lines.push('')
73
+ if (Object.keys(trace).length > 0) {
74
+ lines.push('**Trace**')
75
+ for (const [k, v] of Object.entries(trace)) {
76
+ lines.push(`- ${k}: \`${v}\``)
77
+ }
78
+ lines.push('')
79
+ }
80
+ lines.push('</details>')
81
+ lines.push('')
82
+
83
+ lines.push('<details>')
84
+ lines.push(`<summary>📒 Tests (${report.totals.tests})</summary>`)
85
+ lines.push('')
86
+ lines.push('| Test | Result | Calls | Cost | Heals | Asserts |')
87
+ lines.push('| --- | --- | ---: | ---: | ---: | ---: |')
88
+ for (const t of report.tests) {
89
+ const result = t.ok ? '✅ pass' : '❌ fail'
90
+ lines.push(`| ${t.name} | ${result} | ${t.visionCalls} | ${formatUsd(t.visionCostUsd)} | ${t.healEvents?.length ?? 0} | ${t.asserts?.length ?? 0} |`)
91
+ }
92
+ lines.push('')
93
+ lines.push('</details>')
94
+ lines.push('')
95
+
96
+ lines.push('<details>')
97
+ lines.push('<summary>💰 Cost ledger</summary>')
98
+ lines.push('')
99
+ lines.push('| Line item | Value |')
100
+ lines.push('| --- | ---: |')
101
+ lines.push(`| Vision calls | ${report.totals.visionCalls} |`)
102
+ const perCall =
103
+ report.totals.visionCalls > 0
104
+ ? formatUsd(report.totals.visionCostUsd / report.totals.visionCalls)
105
+ : '$0.00'
106
+ lines.push(`| Per-call cost (avg) | ${perCall} |`)
107
+ for (const model of Object.keys(report.totals.callsByModel ?? {}).sort()) {
108
+ lines.push(`| Calls (${model}) | ${report.totals.callsByModel[model]} |`)
109
+ lines.push(`| Spend (${model}) | ${formatUsd(report.totals.costByModel?.[model] ?? 0)} |`)
110
+ }
111
+ lines.push(`| Total vision spend | ${formatUsd(report.totals.visionCostUsd)} |`)
112
+ lines.push(`| Sandbox seconds | ${report.totals.sandboxSeconds.toFixed(1)}s |`)
113
+ if (budgetCap > 0) {
114
+ lines.push(`| Budget cap | ${formatUsd(budgetCap)} |`)
115
+ lines.push(`| Budget exceeded | ${report.totals.budgetExceeded ? '⚠️ yes' : '✅ no'} |`)
116
+ }
117
+ lines.push('')
118
+ lines.push('</details>')
119
+ lines.push('')
120
+
121
+ lines.push('<details>')
122
+ lines.push('<summary>🔧 Heal events</summary>')
123
+ lines.push('')
124
+ const heals = report.tests.flatMap((t) => t.healEvents ?? [])
125
+ if (heals.length === 0) {
126
+ lines.push('No heals this run.')
127
+ } else {
128
+ for (const h of heals) {
129
+ lines.push(`- \`${h.instruction}\` healed with ${h.model || 'unknown model'}`)
130
+ }
131
+ }
132
+ lines.push('')
133
+ lines.push('</details>')
134
+ lines.push('')
135
+
136
+ lines.push('<details>')
137
+ lines.push('<summary>✅ Assertions</summary>')
138
+ lines.push('')
139
+ let any = false
140
+ for (const t of report.tests) {
141
+ if (!t.asserts || t.asserts.length === 0) continue
142
+ any = true
143
+ lines.push(`**${t.name}**`)
144
+ for (const a of t.asserts) {
145
+ const icon = a.verdict === 'pass' ? '✅' : a.verdict === 'fail' ? '❌' : '⚪'
146
+ lines.push(`- ${icon} *${a.question}* — ${a.reasoning}`)
147
+ }
148
+ lines.push('')
149
+ }
150
+ if (!any) {
151
+ lines.push('No assertions recorded.')
152
+ lines.push('')
153
+ }
154
+ lines.push('</details>')
155
+ lines.push('')
156
+
157
+ lines.push('<details>')
158
+ lines.push('<summary>📂 Evidence</summary>')
159
+ lines.push('')
160
+ if (report.artifacts && report.artifacts.videos.length > 0) {
161
+ for (const v of report.artifacts.videos) lines.push(`- video: \`${v}\``)
162
+ }
163
+ if (runUrl) lines.push(`- [workflow run / artifacts](${runUrl})`)
164
+ if ((!report.artifacts || report.artifacts.videos.length === 0) && !runUrl) {
165
+ lines.push('No artifact links available.')
166
+ }
167
+ lines.push('')
168
+ lines.push('</details>')
169
+ lines.push('')
170
+
171
+ lines.push('<details>')
172
+ lines.push('<summary>🚥 Pre-merge checks</summary>')
173
+ lines.push('')
174
+ lines.push('| Check | Status | Explanation |')
175
+ lines.push('| --- | --- | --- |')
176
+ lines.push(`| Tests | ${report.ok ? '✅ Passed' : '❌ Failed'} | ${report.totals.passed}/${report.totals.tests} tests passed |`)
177
+ lines.push(`| Budget | ${report.totals.budgetExceeded ? '⚠️ Warning' : '✅ Passed'} | ${formatUsd(report.totals.visionCostUsd)} spent${budgetCap > 0 ? ` of ${formatUsd(budgetCap)}` : ''} |`)
178
+ lines.push(`| Heal events | ${healCount === 0 ? '✅ Passed' : '⚠️ Warning'} | ${healCount} heal event${healCount === 1 ? '' : 's'} |`)
179
+ lines.push(`| Assertions | ${assertFails === 0 ? '✅ Passed' : '❌ Failed'} | ${assertFails === 0 ? assertCount : `${assertFails} failed`} assertion${assertCount === 1 ? '' : 's'} |`)
180
+ lines.push(`| OpenRouter key | ✅ Passed | \`OPENROUTER_API_KEY\` configured |`)
181
+ if (codeReview && !codeReview.skipped) {
182
+ const codeStatus = codeReview.ok ? '✅ Passed' : '❌ Failed'
183
+ lines.push(`| Code review | ${codeStatus} | ${codeReview.findings.length} findings (${codeReview.model}) |`)
184
+ } else {
185
+ lines.push(`| Code review | ⚪ Skipped | ${codeReview?.summary ?? 'no report'} |`)
186
+ }
187
+ lines.push('')
188
+ lines.push('</details>')
189
+ lines.push('')
190
+
191
+ if (codeReview && !codeReview.skipped) {
192
+ lines.push('<details>')
193
+ lines.push('<summary>🧠 Code review</summary>')
194
+ lines.push('')
195
+ lines.push(`**Verdict:** ${codeReview.verdict} · ${codeReview.model} · ${codeReview.tokens}tok ${formatUsd(codeReview.visionCostUsd)}`)
196
+ lines.push('')
197
+ lines.push(codeReview.summary)
198
+ lines.push('')
199
+ if (codeReview.findings.length > 0) {
200
+ const evidenceIcon = { exercised: '✅', corroborated: '🔴', not_exercised: '⚪', inconclusive: '❔' }
201
+ lines.push('| File | Severity | Evidence | Finding |')
202
+ lines.push('| --- | --- | --- | --- |')
203
+ for (const f of codeReview.findings) {
204
+ const ev = f.evidence
205
+ ? `${evidenceIcon[f.evidence.status] ?? '❔'} ${f.evidence.detail}`
206
+ : '—'
207
+ lines.push(`| \`${f.file}\` | ${f.severity} | ${ev} | ${f.message} |`)
208
+ }
209
+ lines.push('')
210
+ }
211
+ lines.push('</details>')
212
+ lines.push('')
213
+ }
214
+
215
+ lines.push('<details>')
216
+ lines.push('<summary>✨ Actions</summary>')
217
+ lines.push('')
218
+ lines.push('- [ ] Re-run argus-reviewer')
219
+ lines.push('- [ ] Open a heal PR')
220
+ lines.push('- [ ] Record a new flow')
221
+ lines.push('')
222
+ lines.push('</details>')
223
+ lines.push('')
224
+ lines.push('---')
225
+ lines.push('')
226
+ lines.push('<sub>`argus-reviewer` — self-hosted, BYOK OpenRouter UI regression.</sub>')
227
+ lines.push('')
228
+ return lines.join('\n')
229
+ }
230
+
231
+ async function main() {
232
+ const pr = context.payload && context.payload.pull_request
233
+ const owner = context.repo.owner
234
+ const repo = context.repo.repo
235
+ const hasKey = !!process.env.OPENROUTER_API_KEY
236
+ const workDir = process.env.VISION_E2E_WORKING_DIR || ''
237
+ const reportDir = path.resolve(
238
+ process.env.GITHUB_WORKSPACE,
239
+ workDir,
240
+ process.env.ARGUS_REPORT_DIR || 'argus-reviewer-report',
241
+ )
242
+ const runUrl = `${process.env.GITHUB_SERVER_URL}/${owner}/${repo}/actions/runs/${process.env.GITHUB_RUN_ID}`
243
+
244
+ let report
245
+ let codeReview
246
+ if (hasKey) {
247
+ try {
248
+ const raw = fs.readFileSync(path.join(reportDir, 'run.json'), 'utf8')
249
+ report = JSON.parse(raw)
250
+ } catch {
251
+ report = undefined
252
+ }
253
+ try {
254
+ const raw = fs.readFileSync(path.join(reportDir, 'code-review.json'), 'utf8')
255
+ codeReview = JSON.parse(raw)
256
+ } catch {
257
+ codeReview = undefined
258
+ }
259
+ }
260
+
261
+ // Missing code-review.json after a continue-on-error step means the review
262
+ // crashed, not that it skipped — an intentional skip writes ok+skipped.
263
+ // Fail closed rather than reporting it as a clean skip.
264
+ const codeReviewOk = codeReview != null && codeReview.ok === true
265
+ const ok = (report?.ok === true) && codeReviewOk
266
+ const conclusion = !hasKey ? 'neutral' : ok ? 'success' : 'failure'
267
+ const body = !hasKey
268
+ ? renderMissingKeyBody()
269
+ : report === undefined
270
+ ? renderNoReportBody(reportDir, runUrl)
271
+ : renderBody(report, codeReview, runUrl, ok)
272
+
273
+ async function postInlineComments(pr, codeReview) {
274
+ if (!pr || !codeReview || codeReview.skipped || !codeReview.findings) return
275
+ // Must match the severity vocabulary emitted by the code-review schema
276
+ // (src/cli.ts): bug/risk are inline-worthy; nit/q stay in the sticky body.
277
+ const inlineSeverities = ['bug', 'risk']
278
+ const comments = codeReview.findings
279
+ .filter((f) => f.file && typeof f.line === 'number' && inlineSeverities.includes(f.severity))
280
+ .map((f) => ({
281
+ path: f.file,
282
+ line: f.line,
283
+ side: 'RIGHT',
284
+ body: `**argus-reviewer ${f.severity}:** ${f.message}${f.evidence && f.evidence.status !== 'exercised' ? `\n\n*CI evidence: ${f.evidence.detail}*` : ''}`,
285
+ }))
286
+ if (comments.length === 0) return
287
+
288
+ // Re-runs on the same SHA must not duplicate inline comments — the sticky
289
+ // body is upserted but review comments are not. Paginate fully (100/page)
290
+ // and scope dedup to the current head: comments on older commits must not
291
+ // suppress findings that still apply to this head.
292
+ const posted = new Set()
293
+ let page = 1
294
+ for (;;) {
295
+ const { data: existing } = await github.rest.pulls.listReviewComments({
296
+ owner: context.repo.owner,
297
+ repo: context.repo.repo,
298
+ pull_number: pr.number,
299
+ per_page: 100,
300
+ page,
301
+ })
302
+ for (const c of existing) {
303
+ if (c.body && c.body.startsWith('**argus-reviewer') && c.commit_id === pr.head.sha) {
304
+ posted.add(`${c.path}:${c.line}:${c.body}`)
305
+ }
306
+ }
307
+ if (existing.length < 100) break
308
+ page += 1
309
+ }
310
+ const fresh = comments.filter((c) => !posted.has(`${c.path}:${c.line}:${c.body}`))
311
+ if (fresh.length === 0) return
312
+
313
+ // One batched review instead of N createReviewComment calls — avoids
314
+ // secondary rate limits on large findings sets.
315
+ try {
316
+ await github.rest.pulls.createReview({
317
+ owner: context.repo.owner,
318
+ repo: context.repo.repo,
319
+ pull_number: pr.number,
320
+ commit_id: pr.head.sha,
321
+ event: 'COMMENT',
322
+ comments: fresh,
323
+ })
324
+ } catch (e) {
325
+ core.warning(`inline review failed: ${e.message}`)
326
+ }
327
+ }
328
+
329
+ if (pr) {
330
+ const { data: comments } = await github.rest.issues.listComments({
331
+ owner,
332
+ repo,
333
+ issue_number: pr.number,
334
+ per_page: 100,
335
+ })
336
+ const existing = comments.find((c) => c.body && c.body.includes(SENTINEL))
337
+ if (existing) {
338
+ await github.rest.issues.updateComment({
339
+ owner,
340
+ repo,
341
+ comment_id: existing.id,
342
+ body,
343
+ })
344
+ } else {
345
+ await github.rest.issues.createComment({
346
+ owner,
347
+ repo,
348
+ issue_number: pr.number,
349
+ body,
350
+ })
351
+ }
352
+ await postInlineComments(pr, codeReview)
353
+ }
354
+
355
+ const sha = pr ? pr.head.sha : context.sha
356
+ // Commit statuses have no 'neutral'; a 'pending' skip would wedge a
357
+ // required check forever, so skip maps to success with a clear label.
358
+ const state = conclusion === 'failure' ? 'failure' : 'success'
359
+ const description =
360
+ conclusion === 'neutral'
361
+ ? 'argus-reviewer skipped (no OPENROUTER_API_KEY)'
362
+ : `argus-reviewer ${conclusion}`
363
+ await github.rest.repos.createCommitStatus({
364
+ owner,
365
+ repo,
366
+ sha,
367
+ state,
368
+ description,
369
+ context: 'argus-reviewer',
370
+ target_url: runUrl,
371
+ })
372
+
373
+ core.setOutput('conclusion', conclusion)
374
+ }
375
+
376
+ return await main()