argus-reviewer-e2e 0.3.0 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. package/README.md +84 -71
  2. package/action/action.yml +134 -10
  3. package/action/approval-review.mjs +13 -3
  4. package/action/bootstrap.mjs +4 -0
  5. package/action/emit-review.mjs +16 -0
  6. package/action/runtime.mjs +32 -0
  7. package/action/sticky-comment.cjs +1260 -479
  8. package/dist/cli.d.ts +103 -7
  9. package/dist/cli.js +1208 -186
  10. package/dist/config.d.ts +96 -11
  11. package/dist/config.js +102 -4
  12. package/dist/detect.d.ts +29 -2
  13. package/dist/detect.js +98 -7
  14. package/dist/driver/browser.d.ts +32 -0
  15. package/dist/driver/browser.js +56 -1
  16. package/dist/driver/target.d.ts +4 -1
  17. package/dist/driver/target.js +27 -6
  18. package/dist/engine/actions.d.ts +5 -0
  19. package/dist/engine/actions.js +8 -0
  20. package/dist/engine/explore.d.ts +78 -0
  21. package/dist/engine/explore.js +373 -0
  22. package/dist/engine/loop.d.ts +2 -2
  23. package/dist/engine/loop.js +8 -8
  24. package/dist/engine/prompts.d.ts +28 -1
  25. package/dist/engine/prompts.js +88 -0
  26. package/dist/evidence/ci.d.ts +13 -1
  27. package/dist/evidence/ci.js +38 -3
  28. package/dist/evidence/gate.d.ts +8 -0
  29. package/dist/evidence/gate.js +1 -1
  30. package/dist/evidence/link.js +1 -1
  31. package/dist/executor/a0.d.ts +114 -1
  32. package/dist/executor/a0.js +216 -4
  33. package/dist/fsutil.d.ts +3 -2
  34. package/dist/fsutil.js +7 -4
  35. package/dist/journal/schema.d.ts +1 -1
  36. package/dist/log.d.ts +2 -1
  37. package/dist/log.js +10 -2
  38. package/dist/mention.d.ts +45 -0
  39. package/dist/mention.js +107 -0
  40. package/dist/pipeline/app.d.ts +126 -0
  41. package/dist/pipeline/app.js +250 -0
  42. package/dist/pipeline/budget.d.ts +1 -0
  43. package/dist/pipeline/budget.js +1 -1
  44. package/dist/pipeline/verify.d.ts +20 -3
  45. package/dist/pipeline/verify.js +189 -35
  46. package/dist/probe/persist.d.ts +68 -0
  47. package/dist/probe/persist.js +184 -0
  48. package/dist/probe/queue.d.ts +12 -0
  49. package/dist/probe/queue.js +10 -2
  50. package/dist/report/brand-assets.generated.d.ts +9 -0
  51. package/dist/report/brand-assets.generated.js +8 -0
  52. package/dist/report/comment.d.ts +99 -6
  53. package/dist/report/comment.js +292 -103
  54. package/dist/report/html.d.ts +50 -0
  55. package/dist/report/html.js +879 -0
  56. package/dist/report/manifest.d.ts +29 -0
  57. package/dist/report/manifest.js +37 -0
  58. package/dist/report/run.d.ts +54 -1
  59. package/dist/report/run.js +34 -9
  60. package/dist/report/viewmodel.d.ts +91 -0
  61. package/dist/report/viewmodel.js +241 -0
  62. package/dist/review/adjudicate.d.ts +6 -6
  63. package/dist/review/adjudicate.js +2 -2
  64. package/dist/review/inline.d.ts +44 -0
  65. package/dist/review/inline.js +95 -0
  66. package/dist/review/packs.d.ts +21 -0
  67. package/dist/review/packs.js +47 -0
  68. package/dist/review/scope.d.ts +16 -0
  69. package/dist/review/scope.js +74 -0
  70. package/dist/review/secrets.d.ts +10 -10
  71. package/dist/review/secrets.js +7 -7
  72. package/dist/review/testfiles.d.ts +18 -0
  73. package/dist/review/testfiles.js +26 -0
  74. package/dist/review/triage.d.ts +1 -1
  75. package/dist/review/triage.js +10 -10
  76. package/dist/review/validate.d.ts +41 -0
  77. package/dist/review/validate.js +76 -0
  78. package/dist/ui/errors.d.ts +54 -0
  79. package/dist/ui/errors.js +236 -0
  80. package/dist/ui/style.d.ts +34 -0
  81. package/dist/ui/style.js +48 -0
  82. package/dist/ui/summary.d.ts +38 -0
  83. package/dist/ui/summary.js +101 -0
  84. package/dist/vision/cost.d.ts +1 -1
  85. package/dist/vision/decisions.d.ts +9 -3
  86. package/dist/vision/decisions.js +31 -21
  87. package/dist/vision/openrouter.d.ts +4 -0
  88. package/dist/vision/openrouter.js +30 -4
  89. package/package.json +11 -2
package/README.md CHANGED
@@ -1,7 +1,8 @@
1
- # Argus
2
-
3
1
  <p align="center">
4
- <img src="docs/assets/social.png" alt="Argus — vision-model E2E testing" width="640" />
2
+ <picture>
3
+ <source media="(prefers-color-scheme: dark)" srcset="assets/brand/export/lockup-dark.svg" />
4
+ <img src="assets/brand/export/lockup-light.svg" alt="Argus" width="240" />
5
+ </picture>
5
6
  </p>
6
7
 
7
8
  <p align="center">
@@ -11,108 +12,120 @@
11
12
  <a href="https://github.com/duketopceo/Argus/security/policy"><img src="https://img.shields.io/badge/security-policy-orange" alt="security policy" /></a>
12
13
  </p>
13
14
 
14
- **The hundred-eyed watcher for your pull requests.** Argus reviews your diff, then goes further: it starts your real app, clicks through it like a user, and executes probes against suspected bugs — then posts a verdict on the PR with the exact dollar cost. Self-hosted, MIT-licensed, bring-your-own OpenRouter key. No SaaS middleman, no telemetry, no per-seat pricing.
15
+ **Argus reviews your pull request, runs your real app in a browser, and posts a verdict with its exact cost.**
16
+
17
+ It is a GitHub Action and a CLI. It runs on your infrastructure with your own OpenRouter key: no hosted service, no telemetry, no per-seat pricing. MIT licensed.
15
18
 
16
- ## Get started in 60 seconds
19
+ ## Install in 60 seconds
17
20
 
18
21
  ```bash
19
- npm i -D argus-reviewer-e2e # the npm package; the command it installs is `argus-reviewer`
20
- npx argus-reviewer init # writes config + smoke test + GitHub workflow
22
+ npm i -D argus-reviewer-e2e # the package; the command is argus-reviewer
23
+ npx argus-reviewer init # config, a smoke test, and the PR workflow
21
24
  ```
22
25
 
23
- Add `OPENROUTER_API_KEY` to your environment (and repo secrets for CI). Then:
26
+ <img src="docs/assets/demo/init.gif" width="1100" alt="Terminal: argus-reviewer init writes the config, a smoke test and two workflow files, then checks the environment. The OpenRouter key is reported as not set, Playwright chromium is found, and the default lane is code review." />
27
+
28
+ `init` checks your environment and tells you what is missing. Add `OPENROUTER_API_KEY` to your shell and to the repository secrets, and every pull request gets a review.
29
+
30
+ Record a browser flow once, then replay it on every run:
24
31
 
25
32
  ```bash
26
- npx argus-reviewer record "sign in and open the dashboard" --url http://localhost:3000
27
- npx argus-reviewer run # replays + asserts — free on cache hit
33
+ npx argus-reviewer record "add an item and check out" --url http://localhost:3000
34
+ npx argus-reviewer run
28
35
  ```
29
36
 
30
- That's it. `init` drops a ready-to-run GitHub workflow; every PR from then on gets a review comment with findings, flow results, video evidence, and spend.
37
+ A replay that matches the recorded page makes no model call and costs $0. This cast replays a recorded checkout flow with no API key set at all:
38
+
39
+ <img src="docs/assets/demo/run-cache-hit.gif" width="1100" alt="Terminal: cat shows a test that clicks Place order and asserts the order is confirmed. argus-reviewer run passes it from the cache: passed, 1 of 1 tests, total $0.000000 of a $1.00 budget." />
31
40
 
32
41
  ## What lands on your PR
33
42
 
34
- A sticky comment that updates on every push:
43
+ <picture>
44
+ <source media="(prefers-color-scheme: dark)" srcset="docs/assets/hero-dark.png" />
45
+ <img src="docs/assets/hero-light.png" alt="The Argus sticky comment from a real review run: verdict needs changes, 3 findings with suspected proof and none reproduced, review lane failed, flow lane skipped, metered spend $0.000739." />
46
+ </picture>
35
47
 
36
- - **Verdict** — `APPROVE` / `NEEDS_CHANGES` with findings linked to concrete source lines
37
- - **Inline comments on every severity** — one batched PR review, severity-sorted; each finding can carry a committable `suggestion` block you apply in one click
38
- - **Blocks only on proof** — the review escalates to `REQUEST_CHANGES` only for reproduced or adjudicated blockers; everything else stays advisory. Stale request-changes reviews are dismissed automatically, and `requestChanges: false` keeps it advisory forever
39
- - **Flow results** — which recorded user-journeys passed, healed, or broke
40
- - **Reproduced, not suspected** — opt-in sandbox lane runs authored regression probes; a probe that fails on head and passes on base stamps the finding as *proven*
41
- - **Cost** — every model call metered from OpenRouter's per-call pricing, totaled in dollars
48
+ One sticky comment that updates on every push:
42
49
 
43
- That comment is **not a review**. `require_approving_reviews` reads reviews only, so on a protected branch the verdict alone leaves the gate unsatisfied. Supply `approval-token` — a GitHub App installation token, or a PAT from an account that is not the PR author — and Argus also submits a real review whose event follows the verdict. `github.token` cannot do this; GitHub refuses it outright. A token alone is **not** enough: an `APPROVE` also needs `approval-evidence` naming the test command, and `approval-check` naming a check run that has completed green on the pull request's head commit, produced by the App named in `approval-check-app` (`github-actions` by default). Details, the measured refusals, and the review-discipline rules: [`docs/approval-token.md`](docs/approval-token.md).
50
+ - **Verdict:** approve or needs changes, with each finding linked to a source line.
51
+ - **Inline comments:** one batched review, sorted by severity. A finding can carry a `suggestion` block you apply in one click.
52
+ - **Proof level:** each finding says whether it is suspected or reproduced. Argus requests changes only for reproduced or adjudicated blockers; everything else stays advisory. `requestChanges: false` keeps it advisory always.
53
+ - **Lanes and cost:** what ran, what did not and why, and the dollar cost of the model calls.
44
54
 
45
- <p align="center">
46
- <img src="docs/assets/demo.gif" alt="argus-reviewer run — live vision call, PASS, $0.0005 spend" width="900" />
47
- </p>
55
+ The comment is not a GitHub review, so on its own it does not satisfy a required-approval rule. To have Argus submit a real review, supply `approval-token` (a GitHub App token, or a PAT from an account that is not the PR author). An approve also needs `approval-evidence` and a green `approval-check` on the head commit. Details: [`docs/approval-token.md`](docs/approval-token.md).
56
+
57
+ ## Four lanes
58
+
59
+ `argus-reviewer verify` runs the lanes you select and writes one `run-manifest.json`, which the PR comment renders.
60
+
61
+ Beside it, verify writes `report.html`: an offline evidence report with the verdict, each lane, findings with suggestions, the flow step timeline, heals and the spend ledger. It is one self-contained file (inline styles, fonts and icons, no network requests), follows your light or dark setting and prints cleanly. It lands at `argus-reviewer-report/report.html` (or your `report-dir`). The action does not upload it; add an upload step for the report directory, and the PR comment footer names the path inside that artifact:
62
+
63
+ ```yaml
64
+ - uses: actions/upload-artifact@v4
65
+ if: always()
66
+ with:
67
+ name: argus-reviewer-report-${{ github.run_id }}-${{ github.run_attempt }}
68
+ path: argus-reviewer-report/
69
+ ```
70
+
71
+ | Lane | Select with | What it does |
72
+ |---|---|---|
73
+ | review | default | Reviews the diff and posts findings and a verdict |
74
+ | flow | `--flow` (action input `run`) | Replays recorded browser flows, cache first |
75
+ | app | `--app --task "..."` plus `--expect-text`, `--expect-url` or `--expect-selector` | Runs one directed task against your live app and checks the expected state |
76
+ | a0 | `--a0` (action input `a0`) | Hands the task to your own Agent Zero host in a sandboxed child environment. A finished delegation reports inconclusive, never passed, because the agent's answer is self-reported |
77
+
78
+ A lane that cannot run says so. Here the review lane has no pull request to read, so it reports skipped with the reason, and the flow lane replays from the cache:
48
79
 
49
- ## Why it's different
80
+ <img src="docs/assets/demo/verify.gif" width="1100" alt="Terminal: argus-reviewer verify --flow. The review lane is skipped because there is no pull request in context; the flow lane passes; total $0.000000 of a $1.00 budget." />
50
81
 
51
- | | Argus |
52
- |---|---|
53
- | **Selectors** | None. A vision model looks at a screenshot and decides where to click. |
54
- | **Maintenance** | Fingerprint cache replays at zero model cost; when the UI drifts, self-healing re-grounds and the heal shows up as a reviewable diff. |
55
- | **Review depth** | Beyond the diff: full-source evidence linkage, CI evidence, executed probes, real browser runs. |
56
- | **Spend** | You pick the models per lane (`model`, `grounding_model`, `code_model`, `escalation_model`) and set `budgetUsd`. Cache hits cost nothing. |
57
- | **Data** | Yours. Keys, journals, videos, and reports stay on your infra. |
82
+ ## Status legend
83
+
84
+ Every lane, in the comment and in the terminal, reports one of six statuses:
85
+
86
+ | Glyph | Status | Meaning |
87
+ |---|---|---|
88
+ | `●` | passed | The lane ran and its checks held |
89
+ | `⊘` | failed | The lane ran and found a problem |
90
+ | `◐` | inconclusive | The lane ran, but its evidence cannot settle the answer |
91
+ | `⊖` | blocked | A policy stopped the lane |
92
+ | `◌` | unavailable | The lane could not run: a missing key, tool or host |
93
+ | `–` | skipped | The lane was not selected, or had nothing to work on |
94
+
95
+ ## Cost
96
+
97
+ Every OpenRouter call is metered from the provider's per-call price and totaled in the comment. The review in the image above cost **$0.000739**. A flow replay that matches its cache costs $0. `budgetUsd` caps each run (default $1.00), and you choose the model for each job (`model`, `code_model`, `escalation_model`).
98
+
99
+ The action tags every call with `ARGUS_REVIEWER_TRACE` (repository, PR, commit, run), so spend can be attributed per review. See [`docs/quickstart.md`](docs/quickstart.md) for the `openrouter` config block.
100
+
101
+ ## Security
102
+
103
+ Argus treats pull request content as hostile. Untrusted checkouts never execute config code, fork PRs are gated behind the `argus-probe` label, secrets are filtered from model input and comment output, and the probe sandbox runs with no network and a read-only filesystem. Threat model: [`SECURITY.md`](SECURITY.md).
58
104
 
59
105
  ## Configuration
60
106
 
61
- `argus-reviewer.config.ts`:
107
+ `argus-reviewer.config.ts` (or `argus-reviewer.config.json`):
62
108
 
63
109
  ```ts
64
110
  import { defineConfig } from 'argus-reviewer-e2e'
65
111
 
66
112
  export default defineConfig({
67
- model: 'google/gemini-2.5-flash-lite', // vision grounding + actions
113
+ model: 'google/gemini-2.5-flash-lite', // vision: grounding and actions
68
114
  code_model: 'deepseek/deepseek-v4.1-flash', // diff review
69
- escalation_model: 'anthropic/claude-sonnet-4', // risky/complex findings
115
+ escalation_model: 'anthropic/claude-sonnet-4', // risky or complex findings
70
116
  budgetUsd: 1.0,
71
117
  target: { url: 'https://your-app.example.com' },
72
118
  testsDir: 'e2e',
119
+ reportRetention: 20, // archived manifests to keep
73
120
  })
74
121
  ```
75
122
 
76
- Point `provider.order` at fast OpenRouter backends (`cerebras`, `groq`) for sub-second review calls — speed is a routing choice, not a pricing tier. Full shape: [`src/config.ts`](src/config.ts) (a legacy `vision-e2e.config.*` is still accepted). Setup walkthrough: [`docs/quickstart.md`](docs/quickstart.md).
123
+ Code review skips generated, fixture and vendored paths by default (`dist/**`, `fixtures/**`, `tests/goldens/**`, lockfiles, `*.generated.*`, `assets/brand/export/**`). Set `review: { exclude: [...] }` to replace that list (`[]` excludes nothing). The sticky comment's Diagnostics fold says how many files were left out.
77
124
 
78
- ## The execution ladder
125
+ Full shape: [`src/config.ts`](src/config.ts). Setup walkthrough: [`docs/quickstart.md`](docs/quickstart.md).
79
126
 
80
- Argus does more as you grant it more access — each rung is opt-in:
127
+ ## Contributing
81
128
 
82
- 1. **API review** — GitHub token only. Reviews the PR diff and posts the verdict.
83
- 2. **Trusted checkout** — findings get verified against the full source tree.
84
- 3. **Browser flows** — Playwright drives your real app through recorded journeys.
85
- 4. **Sandbox probes** — suspected findings get authored regression tests, executed in a hardened container (no network, no secrets, read-only FS). Fork PRs stay behind an `argus-probe` label gate.
86
- 5. **Agent Zero delegation** — `argus-reviewer delegate "find the checkout bug"` hands exploratory work to your own A0 instance.
87
-
88
- ## Cost attribution
89
-
90
- Every OpenRouter call carries a trace tag. The action auto-sets `ARGUS_REVIEWER_TRACE` (repo, PR, commit, run id) so spend is attributable per review — or set it yourself for custom fields. See [`docs/quickstart.md`](docs/quickstart.md) for the `openrouter` config block.
91
-
92
- ## Security model
93
-
94
- Reviews run against hostile input by design: untrusted checkouts never execute config code, fork PRs are label-gated, secrets are filtered from model context and comment output, and the sandbox probe lane runs network-less with a read-only filesystem. Threat model: [`SECURITY.md`](SECURITY.md).
95
-
96
- ## File structure
97
-
98
- ```text
99
- argus-reviewer/
100
- ├── action/ # GitHub Actions composite action + sticky PR comment
101
- ├── runner/ # Self-hosted runner registration docs + script
102
- ├── electron/ # Local observability dashboard (`npm run app`)
103
- ├── src/
104
- │ ├── api.ts # Test-facing `test`/`td` API + generated test renderer
105
- │ ├── cli.ts # record · run · code-review · delegate · cache · index · init
106
- │ ├── config.ts # `argus-reviewer.config.*` loader
107
- │ ├── cache/ # Per-step fingerprint + flow store
108
- │ ├── driver/ # Playwright browser + dev-server target
109
- │ ├── engine/ # Vision record/replay + healing loop
110
- │ ├── evidence/ # PR/CI context, fork trust gate, finding linkage
111
- │ ├── executor/ # Agent Zero delegation + hardened probe sandbox
112
- │ ├── probe/ # Model-authored regression probes
113
- │ ├── report/ # PR comment, JUnit XML, run.json
114
- │ └── vision/ # OpenRouter client, cost parsing, budget ledger
115
- └── tests/ # Unit tests + Playwright fixture page
116
- ```
129
+ The repository also has a terminal view (`npm run watch`) and a desktop dashboard (`npm run app`). They are contributor tools: they run only from a clone of this repository, and the npm package does not ship them. The casts above are recorded by `npm run demo:record` (see `scripts/demo-record.mjs`). Guidelines: [`CONTRIBUTING.md`](CONTRIBUTING.md).
117
130
 
118
131
  License: [MIT](LICENSE).
package/action/action.yml CHANGED
@@ -1,6 +1,9 @@
1
1
  name: argus-reviewer
2
2
  description: Run the argus-reviewer harness and post a sticky PR comment with cost and evidence.
3
3
  author: duketopceo
4
+ branding:
5
+ icon: eye
6
+ color: blue
4
7
  inputs:
5
8
  openrouter-api-key:
6
9
  description: OpenRouter API key (BYOK). All vision calls are billed through this key.
@@ -44,13 +47,21 @@ inputs:
44
47
  default: argus-reviewer-report
45
48
  required: false
46
49
  sandbox:
47
- description: Enable the B.2 probe lane — authored test probes for unexercised findings run in a hardened Docker container (no network, no secrets, read-only fs). Enable-only — 'false' does not override a config-enabled lane. Requires Docker on the runner; forced off on pull_request_target.
50
+ description: Enable the B.2 probe lane: authored test probes for unexercised findings run in a hardened Docker container (no network, no secrets, read-only fs). Enable-only: 'false' does not override a config-enabled lane. Requires Docker on the runner; forced off on pull_request_target.
48
51
  default: 'false'
49
52
  required: false
50
53
  max-comments:
51
54
  description: Cap on inline review comments posted per run; overflow is summarized in the sticky. Overrides review.maxComments when set.
52
55
  default: ''
53
56
  required: false
57
+ code-model:
58
+ description: Optional OpenRouter model slug for code review (e.g. openai/gpt-5-mini). Wins over the checkout's config code_model, including on pull_request events, where PR config never executes.
59
+ default: ''
60
+ required: false
61
+ review-profiles:
62
+ description: Optional comma-separated review lenses appended to the review prompt: security, perf, debloat (e.g. 'security,perf'). Wins over the checkout's config review.profiles.
63
+ default: ''
64
+ required: false
54
65
  install-consumer-dependencies:
55
66
  description: Explicitly install the consumer project's dependencies for a trusted runtime lane. Disabled by default so code review never runs PR-controlled lifecycle scripts beside secrets.
56
67
  default: 'false'
@@ -59,6 +70,45 @@ inputs:
59
70
  description: Run the browser-flow lane (`argus-reviewer run`). Disabled by default for public code-review-only repositories. Requires an explicit target and a trusted runtime path.
60
71
  default: 'false'
61
72
  required: false
73
+ app:
74
+ description: >-
75
+ Enable the application task lane (`verify --app`): a directed task
76
+ plus an expected-state check against your running app, never free
77
+ exploration. Requires `app-task`; `app-url` wins over the config's
78
+ target.url. Executable lane: refused on fork PRs and
79
+ pull_request_target by the runtime-lane gate.
80
+ default: 'false'
81
+ required: false
82
+ app-task:
83
+ description: Directed task for the app lane (and the a0 lane), e.g. 'submit the signup form and land on /welcome'. Required when app is true.
84
+ default: ''
85
+ required: false
86
+ app-url:
87
+ description: URL the flow/app/a0 lanes target; wins over the config's target.url. Passed as --url.
88
+ default: ''
89
+ required: false
90
+ app-expect-text:
91
+ description: Text that must appear for the app task to pass (maps to --expect-text).
92
+ default: ''
93
+ required: false
94
+ app-expect-url:
95
+ description: Regular expression the page URL must match after the task (maps to --expect-url).
96
+ default: ''
97
+ required: false
98
+ app-expect-selector:
99
+ description: CSS selector that must be present after the task (maps to --expect-selector).
100
+ default: ''
101
+ required: false
102
+ a0:
103
+ description: >-
104
+ Opt-in Agent Zero escalation lane (`verify --a0`): delegates the
105
+ configured task to your self-hosted A0 host inside a sanitized,
106
+ allowlisted child environment. The lane reports `inconclusive` even
107
+ on agent-reported success (live verification is not proven, #53),
108
+ and `unavailable` when the host/CLI is missing. Bounded to 1 task
109
+ and the configured timeout.
110
+ default: 'false'
111
+ required: false
62
112
  approval-token:
63
113
  description: >-
64
114
  Optional token that lets Argus submit a *formal* pull request review, which
@@ -73,7 +123,7 @@ inputs:
73
123
  approval-evidence:
74
124
  description: >-
75
125
  The test command(s) an approval stands on, cited verbatim in the review
76
- body — e.g. 'python -m unittest discover -s tests && ruff check .'.
126
+ body, e.g. 'python -m unittest discover -s tests && ruff check .'.
77
127
  Required whenever approval-token is supplied: the lane refuses to submit
78
128
  an approval that cites no command, so an approver is always re-runnable.
79
129
  Ignored when approval-token is empty.
@@ -138,26 +188,60 @@ runs:
138
188
  ARGUS_BROWSER: ${{ inputs.browser }}
139
189
  ARGUS_BUDGET_USD: ${{ inputs.budget-usd }}
140
190
  ARGUS_MAX_COMMENTS: ${{ inputs.max-comments }}
191
+ ARGUS_CODE_MODEL: ${{ inputs.code-model }}
192
+ ARGUS_REVIEW_PROFILES: ${{ inputs.review-profiles }}
141
193
  run: node "$ARGUS_ACTION_PATH/bootstrap.mjs"
142
194
 
143
195
  - name: Reject executable lanes on untrusted pull requests
144
- if: inputs.run != 'false' || inputs.install-consumer-dependencies == 'true'
196
+ if: >-
197
+ inputs.run != 'false' || inputs.app == 'true' || inputs.a0 == 'true'
198
+ || inputs.install-consumer-dependencies == 'true'
145
199
  shell: bash
146
200
  env:
147
201
  EVENT_NAME: ${{ github.event_name }}
148
202
  HEAD_FORK: ${{ github.event.pull_request.head.repo.fork }}
149
203
  run: |
150
- if [ "$EVENT_NAME" = "pull_request_target" ]; then
151
- echo "Argus runtime lanes are disabled for pull_request_target." >&2
204
+ # Each refusal says what happened and what to do, in the log and in
205
+ # the job summary, because no PR comment is posted on this path.
206
+ refuse() {
207
+ echo "$1" >&2
208
+ {
209
+ echo '### Argus: ⊖ blocked'
210
+ echo
211
+ echo "**Runtime lanes refused:** $1"
212
+ echo
213
+ echo "Fix: $2"
214
+ echo
215
+ } >> "$GITHUB_STEP_SUMMARY"
152
216
  exit 1
217
+ }
218
+ REVIEW_ONLY="set run: 'false' and leave app, a0 and install-consumer-dependencies off; the code review lane still runs."
219
+ if [ "$EVENT_NAME" = "pull_request_target" ]; then
220
+ refuse "Argus runtime lanes do not run on pull_request_target, which carries write credentials beside PR code." \
221
+ "use the pull_request event for runtime lanes, or $REVIEW_ONLY"
153
222
  fi
154
223
  if [ "$EVENT_NAME" = "pull_request" ] && [ "$HEAD_FORK" != "false" ]; then
155
- echo "Argus runtime lanes require a same-repository PR or an explicitly trusted workflow." >&2
156
- exit 1
224
+ refuse "this pull request comes from a fork, and runtime lanes execute PR code, so they need a same-repository PR or an explicitly trusted workflow." \
225
+ "for fork pull requests, $REVIEW_ONLY"
226
+ fi
227
+ if [ "$EVENT_NAME" = "issue_comment" ]; then
228
+ # issue_comment checks out the BASE ref, never the PR head;
229
+ # consumer deps on the base tree are trusted. The replay `run`
230
+ # lane is still meaningless here (it would test the base app
231
+ # against a head diff); @argus record covers that case.
232
+ if [ "${{ inputs.run }}" != "false" ]; then
233
+ refuse "the run lane is disabled on issue_comment, which checks out the base branch, not the PR head." \
234
+ "comment '@argus record' on the pull request instead."
235
+ fi
236
+ if [ "${{ inputs.app }}" = "true" ] || [ "${{ inputs.a0 }}" = "true" ]; then
237
+ refuse "the app and a0 lanes are disabled on issue_comment; they verify a checkout of the PR head." \
238
+ "run those lanes from a pull_request workflow."
239
+ fi
240
+ exit 0
157
241
  fi
158
242
  if [ "$EVENT_NAME" != "pull_request" ] && [ "$EVENT_NAME" != "" ]; then
159
- echo "Argus runtime lanes are disabled for unlisted CI events: $EVENT_NAME" >&2
160
- exit 1
243
+ refuse "runtime lanes do not run on the $EVENT_NAME event, which is not on the trusted list." \
244
+ "run them from a pull_request workflow, or $REVIEW_ONLY"
161
245
  fi
162
246
 
163
247
  - name: Install consumer dependencies for explicit runtime lanes
@@ -167,7 +251,7 @@ runs:
167
251
  run: npm ci
168
252
 
169
253
  - name: Install Playwright browser for explicit runtime lanes
170
- if: inputs.run != 'false'
254
+ if: inputs.run != 'false' || inputs.app == 'true' || (github.event_name == 'issue_comment' && contains(github.event.comment.body, 'record'))
171
255
  shell: bash
172
256
  working-directory: ${{ inputs.working-directory || '.' }}
173
257
  env:
@@ -186,7 +270,38 @@ runs:
186
270
  ARGUS_WORKING_DIRECTORY: ${{ steps.bootstrap.outputs.working-directory }}
187
271
  run: node "$ARGUS_ACTION_PATH/cli.mjs" index
188
272
 
273
+ - name: Dispatch @argus mention
274
+ # issue_comment runs never check out the PR head; the workflow
275
+ # checks out the base ref and `mention` dispatches whitelisted
276
+ # commands (review/record/persist/help). Results still land in the
277
+ # report dir for the sticky post step below.
278
+ if: github.event_name == 'issue_comment'
279
+ id: mention
280
+ shell: bash
281
+ working-directory: ${{ inputs.working-directory || '.' }}
282
+ env:
283
+ ARGUS_ACTION_PATH: ${{ github.action_path }}
284
+ OPENROUTER_API_KEY: ${{ inputs.openrouter-api-key }}
285
+ GITHUB_TOKEN: ${{ github.token }}
286
+ ARGUS_CLI_JSON: ${{ steps.bootstrap.outputs.cli-json }}
287
+ ARGUS_WORKING_DIRECTORY: ${{ steps.bootstrap.outputs.working-directory }}
288
+ ARGUS_REPORT_DIR: ${{ inputs.report-dir }}
289
+ ARGUS_BUDGET_USD: ${{ inputs.budget-usd }}
290
+ ARGUS_MAX_COMMENTS: ${{ inputs.max-comments }}
291
+ ARGUS_CODE_MODEL: ${{ inputs.code-model }}
292
+ ARGUS_REVIEW_PROFILES: ${{ inputs.review-profiles }}
293
+ ARGUS_DEBUG: '1'
294
+ ARGUS_REVIEWER_TRACE: >-
295
+ {"repo":"${{ github.repository }}",
296
+ "pr":"${{ github.event.issue.number }}",
297
+ "commit":"${{ github.sha }}",
298
+ "run_id":"${{ github.run_id }}",
299
+ "run_attempt":"${{ github.run_attempt }}",
300
+ "workflow":"${{ github.workflow }}"}
301
+ run: node "$ARGUS_ACTION_PATH/cli.mjs" mention --report-dir "$ARGUS_REPORT_DIR"
302
+
189
303
  - name: Run selected Argus lanes
304
+ if: github.event_name != 'issue_comment'
190
305
  id: verify
191
306
  shell: bash
192
307
  working-directory: ${{ inputs.working-directory || '.' }}
@@ -199,11 +314,20 @@ runs:
199
314
  ARGUS_WORKING_DIRECTORY: ${{ steps.bootstrap.outputs.working-directory }}
200
315
  ARGUS_REPORT_DIR: ${{ inputs.report-dir }}
201
316
  ARGUS_VERIFY_FLOW: ${{ inputs.run != 'false' && '1' || '' }}
317
+ ARGUS_VERIFY_APP: ${{ inputs.app == 'true' && '1' || '' }}
318
+ ARGUS_VERIFY_A0: ${{ inputs.a0 == 'true' && '1' || '' }}
319
+ ARGUS_VERIFY_URL: ${{ inputs.app-url }}
320
+ ARGUS_VERIFY_TASK: ${{ inputs.app-task }}
321
+ ARGUS_VERIFY_EXPECT_TEXT: ${{ inputs.app-expect-text }}
322
+ ARGUS_VERIFY_EXPECT_URL: ${{ inputs.app-expect-url }}
323
+ ARGUS_VERIFY_EXPECT_SELECTOR: ${{ inputs.app-expect-selector }}
202
324
  ARGUS_SANDBOX: ${{ inputs.sandbox == 'true' && '1' || '' }}
203
325
  ARGUS_MAX_COMMENTS: ${{ inputs.max-comments }}
204
326
  ARGUS_DEBUG: '1'
205
327
  ARGUS_DIFF_BASE: ${{ inputs.diff-base }}
206
328
  ARGUS_BUDGET_USD: ${{ inputs.budget-usd }}
329
+ ARGUS_CODE_MODEL: ${{ inputs.code-model }}
330
+ ARGUS_REVIEW_PROFILES: ${{ inputs.review-profiles }}
207
331
  ARGUS_REVIEWER_TRACE: >-
208
332
  {"repo":"${{ github.repository }}",
209
333
  "pr":"${{ github.event.pull_request.number }}",
@@ -20,6 +20,12 @@ import { execFile } from 'node:child_process'
20
20
  import { readFile } from 'node:fs/promises'
21
21
  import { join } from 'node:path'
22
22
 
23
+ // The Ocellus verdict vocabulary has one action-side copy (KTD2), in the
24
+ // sticky-comment module; this lane reuses it rather than keeping a third.
25
+ import sticky from './sticky-comment.cjs'
26
+
27
+ const { STATUS_GLYPH, VERDICT_STATUS, VERDICT_LABEL } = sticky
28
+
23
29
  const GH_API = 'https://api.github.com'
24
30
 
25
31
  /** Verdicts that mean "ship it". Everything else blocks. */
@@ -85,12 +91,16 @@ export function classifyApprovalFailure(status, message) {
85
91
  export function reviewBody(codeReview, runUrl, evidence, evidenceCheck) {
86
92
  const lines = []
87
93
  const verdict = String(codeReview?.verdict ?? 'unknown')
88
- lines.push(`**Argus verdict:** \`${verdict}\``)
94
+ lines.push(
95
+ Object.hasOwn(VERDICT_LABEL, verdict)
96
+ ? `**Argus: ${STATUS_GLYPH[VERDICT_STATUS[verdict]]} ${VERDICT_LABEL[verdict]}**`
97
+ : `**Argus:** verdict \`${oneLine(verdict).replace(/`/g, '')}\``,
98
+ )
89
99
  if (codeReview?.summary) lines.push('', codeReview.summary)
90
100
  if (evidence !== undefined && evidence !== '') {
91
101
  lines.push('', `**Verified by:** \`${oneLine(evidence)}\``)
92
102
  } else {
93
- lines.push('', '**Verified by:** none — this approval cites no test command.')
103
+ lines.push('', '**Verified by:** none. This approval cites no test command.')
94
104
  }
95
105
  if (evidenceCheck) {
96
106
  // The command above is the caller's claim; this line is the receipt. A
@@ -113,7 +123,7 @@ export function reviewBody(codeReview, runUrl, evidence, evidenceCheck) {
113
123
  )
114
124
  for (const f of blocking.slice(0, 5)) {
115
125
  const where = f.file ? ` \`${f.file}${f.line ? `:${f.line}` : ''}\`` : ''
116
- lines.push(`- **${f.severity}**${where} — ${oneLine(f.message)}`)
126
+ lines.push(`- **${f.severity}**${where}: ${oneLine(f.message)}`)
117
127
  }
118
128
  }
119
129
  if (runUrl) lines.push('', `[argus-reviewer run](${runUrl})`)
@@ -13,8 +13,10 @@ import {
13
13
  setActionOutput,
14
14
  validateBrowser,
15
15
  validateBudget,
16
+ validateCodeModel,
16
17
  validateMaxComments,
17
18
  validatePathInput,
19
+ validateReviewProfiles,
18
20
  validateVersion,
19
21
  } from './runtime.mjs'
20
22
 
@@ -29,6 +31,8 @@ const reportDir = validatePathInput(
29
31
  validateBrowser(process.env.ARGUS_BROWSER || 'chromium')
30
32
  validateBudget(process.env.ARGUS_BUDGET_USD)
31
33
  validateMaxComments(process.env.ARGUS_MAX_COMMENTS)
34
+ validateCodeModel(process.env.ARGUS_CODE_MODEL)
35
+ validateReviewProfiles(process.env.ARGUS_REVIEW_PROFILES)
32
36
 
33
37
  async function stageConfig() {
34
38
  if (configInput === '' || configInput === 'argus-reviewer.config.ts') return
@@ -219,6 +219,22 @@ export async function emitApprovalReview(env, deps = {}) {
219
219
  return { ok: false, reviewEvent: 'none', reviewState: 'no-report', message }
220
220
  }
221
221
 
222
+ // The report must be this run's own before it can drive any review event.
223
+ // Head sha alone is forgeable (a commit author knows it); GITHUB_RUN_ID is
224
+ // not knowable when a commit or planted file is authored, so a report that
225
+ // cannot present it is residue or plant. Local runs have no run id — the
226
+ // gate is off by design there, same as the sticky post step.
227
+ const expectedNonce = (env.GITHUB_RUN_ID ?? '').trim()
228
+ if (expectedNonce !== '' && codeReview?.runNonce !== expectedNonce) {
229
+ const message =
230
+ 'argus-reviewer: code-review.json does not carry this workflow run\'s id — ' +
231
+ 'it is stale or was not produced by this run. No review was submitted.'
232
+ fail(message)
233
+ setActionOutput('review-event', 'none')
234
+ setActionOutput('review-state', 'stale-report')
235
+ return { ok: false, reviewEvent: 'none', reviewState: 'stale-report', message }
236
+ }
237
+
222
238
  const event = reviewEventFor(codeReview.verdict)
223
239
 
224
240
  // The run reviewed the code at the head SHA its event carried. If the head has
@@ -93,6 +93,38 @@ export function validateMaxComments(raw) {
93
93
  return Number(raw)
94
94
  }
95
95
 
96
+ const KNOWN_REVIEW_PROFILES = new Set(['security', 'perf', 'debloat'])
97
+
98
+ export function validateReviewProfiles(raw) {
99
+ if (raw === undefined || raw === '') return undefined
100
+ const names = String(raw)
101
+ .split(',')
102
+ .map((s) => s.trim())
103
+ .filter((s) => s !== '')
104
+ // Env-only value — no shell — but reject unknown names so a typo fails
105
+ // here instead of silently dropping the intended lens at config load.
106
+ for (const name of names) {
107
+ if (!KNOWN_REVIEW_PROFILES.has(name)) {
108
+ throw new Error(
109
+ `review-profiles must be a comma-separated list of: ${[...KNOWN_REVIEW_PROFILES].join(', ')} — got '${name}'`,
110
+ )
111
+ }
112
+ }
113
+ return names.join(',')
114
+ }
115
+
116
+ export function validateCodeModel(raw) {
117
+ if (raw === undefined || raw === '') return undefined
118
+ const value = String(raw).trim()
119
+ // OpenRouter slugs are `vendor/model` with an optional `:variant`
120
+ // (`deepseek/deepseek-r1:free`). Env-only value — no shell — but keep it
121
+ // to slug-shaped input so typos surface here instead of at the API.
122
+ if (!/^[A-Za-z0-9][A-Za-z0-9._:/-]{0,199}$/.test(value)) {
123
+ throw new Error('code-model must be an OpenRouter model slug (e.g. openai/gpt-5-mini)')
124
+ }
125
+ return value
126
+ }
127
+
96
128
  export function resolveWorkingDirectory(workspace, input) {
97
129
  const root = realpathSync(resolve(workspace))
98
130
  const requested = String(input ?? '').trim()