argus-reviewer-e2e 0.3.0 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +84 -71
- package/action/action.yml +134 -10
- package/action/approval-review.mjs +13 -3
- package/action/bootstrap.mjs +4 -0
- package/action/emit-review.mjs +16 -0
- package/action/runtime.mjs +32 -0
- package/action/sticky-comment.cjs +1260 -479
- package/dist/cli.d.ts +103 -7
- package/dist/cli.js +1208 -186
- package/dist/config.d.ts +96 -11
- package/dist/config.js +102 -4
- package/dist/detect.d.ts +29 -2
- package/dist/detect.js +98 -7
- package/dist/driver/browser.d.ts +32 -0
- package/dist/driver/browser.js +56 -1
- package/dist/driver/target.d.ts +4 -1
- package/dist/driver/target.js +27 -6
- package/dist/engine/actions.d.ts +5 -0
- package/dist/engine/actions.js +8 -0
- package/dist/engine/explore.d.ts +78 -0
- package/dist/engine/explore.js +373 -0
- package/dist/engine/loop.d.ts +2 -2
- package/dist/engine/loop.js +8 -8
- package/dist/engine/prompts.d.ts +28 -1
- package/dist/engine/prompts.js +88 -0
- package/dist/evidence/ci.d.ts +13 -1
- package/dist/evidence/ci.js +38 -3
- package/dist/evidence/gate.d.ts +8 -0
- package/dist/evidence/gate.js +1 -1
- package/dist/evidence/link.js +1 -1
- package/dist/executor/a0.d.ts +114 -1
- package/dist/executor/a0.js +216 -4
- package/dist/fsutil.d.ts +3 -2
- package/dist/fsutil.js +7 -4
- package/dist/journal/schema.d.ts +1 -1
- package/dist/log.d.ts +2 -1
- package/dist/log.js +10 -2
- package/dist/mention.d.ts +45 -0
- package/dist/mention.js +107 -0
- package/dist/pipeline/app.d.ts +126 -0
- package/dist/pipeline/app.js +250 -0
- package/dist/pipeline/budget.d.ts +1 -0
- package/dist/pipeline/budget.js +1 -1
- package/dist/pipeline/verify.d.ts +20 -3
- package/dist/pipeline/verify.js +189 -35
- package/dist/probe/persist.d.ts +68 -0
- package/dist/probe/persist.js +184 -0
- package/dist/probe/queue.d.ts +12 -0
- package/dist/probe/queue.js +10 -2
- package/dist/report/brand-assets.generated.d.ts +9 -0
- package/dist/report/brand-assets.generated.js +8 -0
- package/dist/report/comment.d.ts +99 -6
- package/dist/report/comment.js +292 -103
- package/dist/report/html.d.ts +50 -0
- package/dist/report/html.js +879 -0
- package/dist/report/manifest.d.ts +29 -0
- package/dist/report/manifest.js +37 -0
- package/dist/report/run.d.ts +54 -1
- package/dist/report/run.js +34 -9
- package/dist/report/viewmodel.d.ts +91 -0
- package/dist/report/viewmodel.js +241 -0
- package/dist/review/adjudicate.d.ts +6 -6
- package/dist/review/adjudicate.js +2 -2
- package/dist/review/inline.d.ts +44 -0
- package/dist/review/inline.js +95 -0
- package/dist/review/packs.d.ts +21 -0
- package/dist/review/packs.js +47 -0
- package/dist/review/scope.d.ts +16 -0
- package/dist/review/scope.js +74 -0
- package/dist/review/secrets.d.ts +10 -10
- package/dist/review/secrets.js +7 -7
- package/dist/review/testfiles.d.ts +18 -0
- package/dist/review/testfiles.js +26 -0
- package/dist/review/triage.d.ts +1 -1
- package/dist/review/triage.js +10 -10
- package/dist/review/validate.d.ts +41 -0
- package/dist/review/validate.js +76 -0
- package/dist/ui/errors.d.ts +54 -0
- package/dist/ui/errors.js +236 -0
- package/dist/ui/style.d.ts +34 -0
- package/dist/ui/style.js +48 -0
- package/dist/ui/summary.d.ts +38 -0
- package/dist/ui/summary.js +101 -0
- package/dist/vision/cost.d.ts +1 -1
- package/dist/vision/decisions.d.ts +9 -3
- package/dist/vision/decisions.js +31 -21
- package/dist/vision/openrouter.d.ts +4 -0
- package/dist/vision/openrouter.js +30 -4
- package/package.json +11 -2
package/README.md
CHANGED
|
@@ -1,7 +1,8 @@
|
|
|
1
|
-
# Argus
|
|
2
|
-
|
|
3
1
|
<p align="center">
|
|
4
|
-
<
|
|
2
|
+
<picture>
|
|
3
|
+
<source media="(prefers-color-scheme: dark)" srcset="assets/brand/export/lockup-dark.svg" />
|
|
4
|
+
<img src="assets/brand/export/lockup-light.svg" alt="Argus" width="240" />
|
|
5
|
+
</picture>
|
|
5
6
|
</p>
|
|
6
7
|
|
|
7
8
|
<p align="center">
|
|
@@ -11,108 +12,120 @@
|
|
|
11
12
|
<a href="https://github.com/duketopceo/Argus/security/policy"><img src="https://img.shields.io/badge/security-policy-orange" alt="security policy" /></a>
|
|
12
13
|
</p>
|
|
13
14
|
|
|
14
|
-
**
|
|
15
|
+
**Argus reviews your pull request, runs your real app in a browser, and posts a verdict with its exact cost.**
|
|
16
|
+
|
|
17
|
+
It is a GitHub Action and a CLI. It runs on your infrastructure with your own OpenRouter key: no hosted service, no telemetry, no per-seat pricing. MIT licensed.
|
|
15
18
|
|
|
16
|
-
##
|
|
19
|
+
## Install in 60 seconds
|
|
17
20
|
|
|
18
21
|
```bash
|
|
19
|
-
npm i -D argus-reviewer-e2e # the
|
|
20
|
-
npx argus-reviewer init #
|
|
22
|
+
npm i -D argus-reviewer-e2e # the package; the command is argus-reviewer
|
|
23
|
+
npx argus-reviewer init # config, a smoke test, and the PR workflow
|
|
21
24
|
```
|
|
22
25
|
|
|
23
|
-
|
|
26
|
+
<img src="docs/assets/demo/init.gif" width="1100" alt="Terminal: argus-reviewer init writes the config, a smoke test and two workflow files, then checks the environment. The OpenRouter key is reported as not set, Playwright chromium is found, and the default lane is code review." />
|
|
27
|
+
|
|
28
|
+
`init` checks your environment and tells you what is missing. Add `OPENROUTER_API_KEY` to your shell and to the repository secrets, and every pull request gets a review.
|
|
29
|
+
|
|
30
|
+
Record a browser flow once, then replay it on every run:
|
|
24
31
|
|
|
25
32
|
```bash
|
|
26
|
-
npx argus-reviewer record "
|
|
27
|
-
npx argus-reviewer run
|
|
33
|
+
npx argus-reviewer record "add an item and check out" --url http://localhost:3000
|
|
34
|
+
npx argus-reviewer run
|
|
28
35
|
```
|
|
29
36
|
|
|
30
|
-
|
|
37
|
+
A replay that matches the recorded page makes no model call and costs $0. This cast replays a recorded checkout flow with no API key set at all:
|
|
38
|
+
|
|
39
|
+
<img src="docs/assets/demo/run-cache-hit.gif" width="1100" alt="Terminal: cat shows a test that clicks Place order and asserts the order is confirmed. argus-reviewer run passes it from the cache: passed, 1 of 1 tests, total $0.000000 of a $1.00 budget." />
|
|
31
40
|
|
|
32
41
|
## What lands on your PR
|
|
33
42
|
|
|
34
|
-
|
|
43
|
+
<picture>
|
|
44
|
+
<source media="(prefers-color-scheme: dark)" srcset="docs/assets/hero-dark.png" />
|
|
45
|
+
<img src="docs/assets/hero-light.png" alt="The Argus sticky comment from a real review run: verdict needs changes, 3 findings with suspected proof and none reproduced, review lane failed, flow lane skipped, metered spend $0.000739." />
|
|
46
|
+
</picture>
|
|
35
47
|
|
|
36
|
-
|
|
37
|
-
- **Inline comments on every severity** — one batched PR review, severity-sorted; each finding can carry a committable `suggestion` block you apply in one click
|
|
38
|
-
- **Blocks only on proof** — the review escalates to `REQUEST_CHANGES` only for reproduced or adjudicated blockers; everything else stays advisory. Stale request-changes reviews are dismissed automatically, and `requestChanges: false` keeps it advisory forever
|
|
39
|
-
- **Flow results** — which recorded user-journeys passed, healed, or broke
|
|
40
|
-
- **Reproduced, not suspected** — opt-in sandbox lane runs authored regression probes; a probe that fails on head and passes on base stamps the finding as *proven*
|
|
41
|
-
- **Cost** — every model call metered from OpenRouter's per-call pricing, totaled in dollars
|
|
48
|
+
One sticky comment that updates on every push:
|
|
42
49
|
|
|
43
|
-
|
|
50
|
+
- **Verdict:** approve or needs changes, with each finding linked to a source line.
|
|
51
|
+
- **Inline comments:** one batched review, sorted by severity. A finding can carry a `suggestion` block you apply in one click.
|
|
52
|
+
- **Proof level:** each finding says whether it is suspected or reproduced. Argus requests changes only for reproduced or adjudicated blockers; everything else stays advisory. `requestChanges: false` keeps it advisory always.
|
|
53
|
+
- **Lanes and cost:** what ran, what did not and why, and the dollar cost of the model calls.
|
|
44
54
|
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
55
|
+
The comment is not a GitHub review, so on its own it does not satisfy a required-approval rule. To have Argus submit a real review, supply `approval-token` (a GitHub App token, or a PAT from an account that is not the PR author). An approve also needs `approval-evidence` and a green `approval-check` on the head commit. Details: [`docs/approval-token.md`](docs/approval-token.md).
|
|
56
|
+
|
|
57
|
+
## Four lanes
|
|
58
|
+
|
|
59
|
+
`argus-reviewer verify` runs the lanes you select and writes one `run-manifest.json`, which the PR comment renders.
|
|
60
|
+
|
|
61
|
+
Beside it, verify writes `report.html`: an offline evidence report with the verdict, each lane, findings with suggestions, the flow step timeline, heals and the spend ledger. It is one self-contained file (inline styles, fonts and icons, no network requests), follows your light or dark setting and prints cleanly. It lands at `argus-reviewer-report/report.html` (or your `report-dir`). The action does not upload it; add an upload step for the report directory, and the PR comment footer names the path inside that artifact:
|
|
62
|
+
|
|
63
|
+
```yaml
|
|
64
|
+
- uses: actions/upload-artifact@v4
|
|
65
|
+
if: always()
|
|
66
|
+
with:
|
|
67
|
+
name: argus-reviewer-report-${{ github.run_id }}-${{ github.run_attempt }}
|
|
68
|
+
path: argus-reviewer-report/
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
| Lane | Select with | What it does |
|
|
72
|
+
|---|---|---|
|
|
73
|
+
| review | default | Reviews the diff and posts findings and a verdict |
|
|
74
|
+
| flow | `--flow` (action input `run`) | Replays recorded browser flows, cache first |
|
|
75
|
+
| app | `--app --task "..."` plus `--expect-text`, `--expect-url` or `--expect-selector` | Runs one directed task against your live app and checks the expected state |
|
|
76
|
+
| a0 | `--a0` (action input `a0`) | Hands the task to your own Agent Zero host in a sandboxed child environment. A finished delegation reports inconclusive, never passed, because the agent's answer is self-reported |
|
|
77
|
+
|
|
78
|
+
A lane that cannot run says so. Here the review lane has no pull request to read, so it reports skipped with the reason, and the flow lane replays from the cache:
|
|
48
79
|
|
|
49
|
-
|
|
80
|
+
<img src="docs/assets/demo/verify.gif" width="1100" alt="Terminal: argus-reviewer verify --flow. The review lane is skipped because there is no pull request in context; the flow lane passes; total $0.000000 of a $1.00 budget." />
|
|
50
81
|
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
|
56
|
-
|
|
57
|
-
|
|
|
82
|
+
## Status legend
|
|
83
|
+
|
|
84
|
+
Every lane, in the comment and in the terminal, reports one of six statuses:
|
|
85
|
+
|
|
86
|
+
| Glyph | Status | Meaning |
|
|
87
|
+
|---|---|---|
|
|
88
|
+
| `●` | passed | The lane ran and its checks held |
|
|
89
|
+
| `⊘` | failed | The lane ran and found a problem |
|
|
90
|
+
| `◐` | inconclusive | The lane ran, but its evidence cannot settle the answer |
|
|
91
|
+
| `⊖` | blocked | A policy stopped the lane |
|
|
92
|
+
| `◌` | unavailable | The lane could not run: a missing key, tool or host |
|
|
93
|
+
| `–` | skipped | The lane was not selected, or had nothing to work on |
|
|
94
|
+
|
|
95
|
+
## Cost
|
|
96
|
+
|
|
97
|
+
Every OpenRouter call is metered from the provider's per-call price and totaled in the comment. The review in the image above cost **$0.000739**. A flow replay that matches its cache costs $0. `budgetUsd` caps each run (default $1.00), and you choose the model for each job (`model`, `code_model`, `escalation_model`).
|
|
98
|
+
|
|
99
|
+
The action tags every call with `ARGUS_REVIEWER_TRACE` (repository, PR, commit, run), so spend can be attributed per review. See [`docs/quickstart.md`](docs/quickstart.md) for the `openrouter` config block.
|
|
100
|
+
|
|
101
|
+
## Security
|
|
102
|
+
|
|
103
|
+
Argus treats pull request content as hostile. Untrusted checkouts never execute config code, fork PRs are gated behind the `argus-probe` label, secrets are filtered from model input and comment output, and the probe sandbox runs with no network and a read-only filesystem. Threat model: [`SECURITY.md`](SECURITY.md).
|
|
58
104
|
|
|
59
105
|
## Configuration
|
|
60
106
|
|
|
61
|
-
`argus-reviewer.config.ts
|
|
107
|
+
`argus-reviewer.config.ts` (or `argus-reviewer.config.json`):
|
|
62
108
|
|
|
63
109
|
```ts
|
|
64
110
|
import { defineConfig } from 'argus-reviewer-e2e'
|
|
65
111
|
|
|
66
112
|
export default defineConfig({
|
|
67
|
-
model: 'google/gemini-2.5-flash-lite', // vision grounding
|
|
113
|
+
model: 'google/gemini-2.5-flash-lite', // vision: grounding and actions
|
|
68
114
|
code_model: 'deepseek/deepseek-v4.1-flash', // diff review
|
|
69
|
-
escalation_model: 'anthropic/claude-sonnet-4', // risky
|
|
115
|
+
escalation_model: 'anthropic/claude-sonnet-4', // risky or complex findings
|
|
70
116
|
budgetUsd: 1.0,
|
|
71
117
|
target: { url: 'https://your-app.example.com' },
|
|
72
118
|
testsDir: 'e2e',
|
|
119
|
+
reportRetention: 20, // archived manifests to keep
|
|
73
120
|
})
|
|
74
121
|
```
|
|
75
122
|
|
|
76
|
-
|
|
123
|
+
Code review skips generated, fixture and vendored paths by default (`dist/**`, `fixtures/**`, `tests/goldens/**`, lockfiles, `*.generated.*`, `assets/brand/export/**`). Set `review: { exclude: [...] }` to replace that list (`[]` excludes nothing). The sticky comment's Diagnostics fold says how many files were left out.
|
|
77
124
|
|
|
78
|
-
|
|
125
|
+
Full shape: [`src/config.ts`](src/config.ts). Setup walkthrough: [`docs/quickstart.md`](docs/quickstart.md).
|
|
79
126
|
|
|
80
|
-
|
|
127
|
+
## Contributing
|
|
81
128
|
|
|
82
|
-
|
|
83
|
-
2. **Trusted checkout** — findings get verified against the full source tree.
|
|
84
|
-
3. **Browser flows** — Playwright drives your real app through recorded journeys.
|
|
85
|
-
4. **Sandbox probes** — suspected findings get authored regression tests, executed in a hardened container (no network, no secrets, read-only FS). Fork PRs stay behind an `argus-probe` label gate.
|
|
86
|
-
5. **Agent Zero delegation** — `argus-reviewer delegate "find the checkout bug"` hands exploratory work to your own A0 instance.
|
|
87
|
-
|
|
88
|
-
## Cost attribution
|
|
89
|
-
|
|
90
|
-
Every OpenRouter call carries a trace tag. The action auto-sets `ARGUS_REVIEWER_TRACE` (repo, PR, commit, run id) so spend is attributable per review — or set it yourself for custom fields. See [`docs/quickstart.md`](docs/quickstart.md) for the `openrouter` config block.
|
|
91
|
-
|
|
92
|
-
## Security model
|
|
93
|
-
|
|
94
|
-
Reviews run against hostile input by design: untrusted checkouts never execute config code, fork PRs are label-gated, secrets are filtered from model context and comment output, and the sandbox probe lane runs network-less with a read-only filesystem. Threat model: [`SECURITY.md`](SECURITY.md).
|
|
95
|
-
|
|
96
|
-
## File structure
|
|
97
|
-
|
|
98
|
-
```text
|
|
99
|
-
argus-reviewer/
|
|
100
|
-
├── action/ # GitHub Actions composite action + sticky PR comment
|
|
101
|
-
├── runner/ # Self-hosted runner registration docs + script
|
|
102
|
-
├── electron/ # Local observability dashboard (`npm run app`)
|
|
103
|
-
├── src/
|
|
104
|
-
│ ├── api.ts # Test-facing `test`/`td` API + generated test renderer
|
|
105
|
-
│ ├── cli.ts # record · run · code-review · delegate · cache · index · init
|
|
106
|
-
│ ├── config.ts # `argus-reviewer.config.*` loader
|
|
107
|
-
│ ├── cache/ # Per-step fingerprint + flow store
|
|
108
|
-
│ ├── driver/ # Playwright browser + dev-server target
|
|
109
|
-
│ ├── engine/ # Vision record/replay + healing loop
|
|
110
|
-
│ ├── evidence/ # PR/CI context, fork trust gate, finding linkage
|
|
111
|
-
│ ├── executor/ # Agent Zero delegation + hardened probe sandbox
|
|
112
|
-
│ ├── probe/ # Model-authored regression probes
|
|
113
|
-
│ ├── report/ # PR comment, JUnit XML, run.json
|
|
114
|
-
│ └── vision/ # OpenRouter client, cost parsing, budget ledger
|
|
115
|
-
└── tests/ # Unit tests + Playwright fixture page
|
|
116
|
-
```
|
|
129
|
+
The repository also has a terminal view (`npm run watch`) and a desktop dashboard (`npm run app`). They are contributor tools: they run only from a clone of this repository, and the npm package does not ship them. The casts above are recorded by `npm run demo:record` (see `scripts/demo-record.mjs`). Guidelines: [`CONTRIBUTING.md`](CONTRIBUTING.md).
|
|
117
130
|
|
|
118
131
|
License: [MIT](LICENSE).
|
package/action/action.yml
CHANGED
|
@@ -1,6 +1,9 @@
|
|
|
1
1
|
name: argus-reviewer
|
|
2
2
|
description: Run the argus-reviewer harness and post a sticky PR comment with cost and evidence.
|
|
3
3
|
author: duketopceo
|
|
4
|
+
branding:
|
|
5
|
+
icon: eye
|
|
6
|
+
color: blue
|
|
4
7
|
inputs:
|
|
5
8
|
openrouter-api-key:
|
|
6
9
|
description: OpenRouter API key (BYOK). All vision calls are billed through this key.
|
|
@@ -44,13 +47,21 @@ inputs:
|
|
|
44
47
|
default: argus-reviewer-report
|
|
45
48
|
required: false
|
|
46
49
|
sandbox:
|
|
47
|
-
description: Enable the B.2 probe lane
|
|
50
|
+
description: Enable the B.2 probe lane: authored test probes for unexercised findings run in a hardened Docker container (no network, no secrets, read-only fs). Enable-only: 'false' does not override a config-enabled lane. Requires Docker on the runner; forced off on pull_request_target.
|
|
48
51
|
default: 'false'
|
|
49
52
|
required: false
|
|
50
53
|
max-comments:
|
|
51
54
|
description: Cap on inline review comments posted per run; overflow is summarized in the sticky. Overrides review.maxComments when set.
|
|
52
55
|
default: ''
|
|
53
56
|
required: false
|
|
57
|
+
code-model:
|
|
58
|
+
description: Optional OpenRouter model slug for code review (e.g. openai/gpt-5-mini). Wins over the checkout's config code_model, including on pull_request events, where PR config never executes.
|
|
59
|
+
default: ''
|
|
60
|
+
required: false
|
|
61
|
+
review-profiles:
|
|
62
|
+
description: Optional comma-separated review lenses appended to the review prompt: security, perf, debloat (e.g. 'security,perf'). Wins over the checkout's config review.profiles.
|
|
63
|
+
default: ''
|
|
64
|
+
required: false
|
|
54
65
|
install-consumer-dependencies:
|
|
55
66
|
description: Explicitly install the consumer project's dependencies for a trusted runtime lane. Disabled by default so code review never runs PR-controlled lifecycle scripts beside secrets.
|
|
56
67
|
default: 'false'
|
|
@@ -59,6 +70,45 @@ inputs:
|
|
|
59
70
|
description: Run the browser-flow lane (`argus-reviewer run`). Disabled by default for public code-review-only repositories. Requires an explicit target and a trusted runtime path.
|
|
60
71
|
default: 'false'
|
|
61
72
|
required: false
|
|
73
|
+
app:
|
|
74
|
+
description: >-
|
|
75
|
+
Enable the application task lane (`verify --app`): a directed task
|
|
76
|
+
plus an expected-state check against your running app, never free
|
|
77
|
+
exploration. Requires `app-task`; `app-url` wins over the config's
|
|
78
|
+
target.url. Executable lane: refused on fork PRs and
|
|
79
|
+
pull_request_target by the runtime-lane gate.
|
|
80
|
+
default: 'false'
|
|
81
|
+
required: false
|
|
82
|
+
app-task:
|
|
83
|
+
description: Directed task for the app lane (and the a0 lane), e.g. 'submit the signup form and land on /welcome'. Required when app is true.
|
|
84
|
+
default: ''
|
|
85
|
+
required: false
|
|
86
|
+
app-url:
|
|
87
|
+
description: URL the flow/app/a0 lanes target; wins over the config's target.url. Passed as --url.
|
|
88
|
+
default: ''
|
|
89
|
+
required: false
|
|
90
|
+
app-expect-text:
|
|
91
|
+
description: Text that must appear for the app task to pass (maps to --expect-text).
|
|
92
|
+
default: ''
|
|
93
|
+
required: false
|
|
94
|
+
app-expect-url:
|
|
95
|
+
description: Regular expression the page URL must match after the task (maps to --expect-url).
|
|
96
|
+
default: ''
|
|
97
|
+
required: false
|
|
98
|
+
app-expect-selector:
|
|
99
|
+
description: CSS selector that must be present after the task (maps to --expect-selector).
|
|
100
|
+
default: ''
|
|
101
|
+
required: false
|
|
102
|
+
a0:
|
|
103
|
+
description: >-
|
|
104
|
+
Opt-in Agent Zero escalation lane (`verify --a0`): delegates the
|
|
105
|
+
configured task to your self-hosted A0 host inside a sanitized,
|
|
106
|
+
allowlisted child environment. The lane reports `inconclusive` even
|
|
107
|
+
on agent-reported success (live verification is not proven, #53),
|
|
108
|
+
and `unavailable` when the host/CLI is missing. Bounded to 1 task
|
|
109
|
+
and the configured timeout.
|
|
110
|
+
default: 'false'
|
|
111
|
+
required: false
|
|
62
112
|
approval-token:
|
|
63
113
|
description: >-
|
|
64
114
|
Optional token that lets Argus submit a *formal* pull request review, which
|
|
@@ -73,7 +123,7 @@ inputs:
|
|
|
73
123
|
approval-evidence:
|
|
74
124
|
description: >-
|
|
75
125
|
The test command(s) an approval stands on, cited verbatim in the review
|
|
76
|
-
body
|
|
126
|
+
body, e.g. 'python -m unittest discover -s tests && ruff check .'.
|
|
77
127
|
Required whenever approval-token is supplied: the lane refuses to submit
|
|
78
128
|
an approval that cites no command, so an approver is always re-runnable.
|
|
79
129
|
Ignored when approval-token is empty.
|
|
@@ -138,26 +188,60 @@ runs:
|
|
|
138
188
|
ARGUS_BROWSER: ${{ inputs.browser }}
|
|
139
189
|
ARGUS_BUDGET_USD: ${{ inputs.budget-usd }}
|
|
140
190
|
ARGUS_MAX_COMMENTS: ${{ inputs.max-comments }}
|
|
191
|
+
ARGUS_CODE_MODEL: ${{ inputs.code-model }}
|
|
192
|
+
ARGUS_REVIEW_PROFILES: ${{ inputs.review-profiles }}
|
|
141
193
|
run: node "$ARGUS_ACTION_PATH/bootstrap.mjs"
|
|
142
194
|
|
|
143
195
|
- name: Reject executable lanes on untrusted pull requests
|
|
144
|
-
if:
|
|
196
|
+
if: >-
|
|
197
|
+
inputs.run != 'false' || inputs.app == 'true' || inputs.a0 == 'true'
|
|
198
|
+
|| inputs.install-consumer-dependencies == 'true'
|
|
145
199
|
shell: bash
|
|
146
200
|
env:
|
|
147
201
|
EVENT_NAME: ${{ github.event_name }}
|
|
148
202
|
HEAD_FORK: ${{ github.event.pull_request.head.repo.fork }}
|
|
149
203
|
run: |
|
|
150
|
-
|
|
151
|
-
|
|
204
|
+
# Each refusal says what happened and what to do, in the log and in
|
|
205
|
+
# the job summary, because no PR comment is posted on this path.
|
|
206
|
+
refuse() {
|
|
207
|
+
echo "$1" >&2
|
|
208
|
+
{
|
|
209
|
+
echo '### Argus: ⊖ blocked'
|
|
210
|
+
echo
|
|
211
|
+
echo "**Runtime lanes refused:** $1"
|
|
212
|
+
echo
|
|
213
|
+
echo "Fix: $2"
|
|
214
|
+
echo
|
|
215
|
+
} >> "$GITHUB_STEP_SUMMARY"
|
|
152
216
|
exit 1
|
|
217
|
+
}
|
|
218
|
+
REVIEW_ONLY="set run: 'false' and leave app, a0 and install-consumer-dependencies off; the code review lane still runs."
|
|
219
|
+
if [ "$EVENT_NAME" = "pull_request_target" ]; then
|
|
220
|
+
refuse "Argus runtime lanes do not run on pull_request_target, which carries write credentials beside PR code." \
|
|
221
|
+
"use the pull_request event for runtime lanes, or $REVIEW_ONLY"
|
|
153
222
|
fi
|
|
154
223
|
if [ "$EVENT_NAME" = "pull_request" ] && [ "$HEAD_FORK" != "false" ]; then
|
|
155
|
-
|
|
156
|
-
|
|
224
|
+
refuse "this pull request comes from a fork, and runtime lanes execute PR code, so they need a same-repository PR or an explicitly trusted workflow." \
|
|
225
|
+
"for fork pull requests, $REVIEW_ONLY"
|
|
226
|
+
fi
|
|
227
|
+
if [ "$EVENT_NAME" = "issue_comment" ]; then
|
|
228
|
+
# issue_comment checks out the BASE ref, never the PR head;
|
|
229
|
+
# consumer deps on the base tree are trusted. The replay `run`
|
|
230
|
+
# lane is still meaningless here (it would test the base app
|
|
231
|
+
# against a head diff); @argus record covers that case.
|
|
232
|
+
if [ "${{ inputs.run }}" != "false" ]; then
|
|
233
|
+
refuse "the run lane is disabled on issue_comment, which checks out the base branch, not the PR head." \
|
|
234
|
+
"comment '@argus record' on the pull request instead."
|
|
235
|
+
fi
|
|
236
|
+
if [ "${{ inputs.app }}" = "true" ] || [ "${{ inputs.a0 }}" = "true" ]; then
|
|
237
|
+
refuse "the app and a0 lanes are disabled on issue_comment; they verify a checkout of the PR head." \
|
|
238
|
+
"run those lanes from a pull_request workflow."
|
|
239
|
+
fi
|
|
240
|
+
exit 0
|
|
157
241
|
fi
|
|
158
242
|
if [ "$EVENT_NAME" != "pull_request" ] && [ "$EVENT_NAME" != "" ]; then
|
|
159
|
-
|
|
160
|
-
|
|
243
|
+
refuse "runtime lanes do not run on the $EVENT_NAME event, which is not on the trusted list." \
|
|
244
|
+
"run them from a pull_request workflow, or $REVIEW_ONLY"
|
|
161
245
|
fi
|
|
162
246
|
|
|
163
247
|
- name: Install consumer dependencies for explicit runtime lanes
|
|
@@ -167,7 +251,7 @@ runs:
|
|
|
167
251
|
run: npm ci
|
|
168
252
|
|
|
169
253
|
- name: Install Playwright browser for explicit runtime lanes
|
|
170
|
-
if: inputs.run != 'false'
|
|
254
|
+
if: inputs.run != 'false' || inputs.app == 'true' || (github.event_name == 'issue_comment' && contains(github.event.comment.body, 'record'))
|
|
171
255
|
shell: bash
|
|
172
256
|
working-directory: ${{ inputs.working-directory || '.' }}
|
|
173
257
|
env:
|
|
@@ -186,7 +270,38 @@ runs:
|
|
|
186
270
|
ARGUS_WORKING_DIRECTORY: ${{ steps.bootstrap.outputs.working-directory }}
|
|
187
271
|
run: node "$ARGUS_ACTION_PATH/cli.mjs" index
|
|
188
272
|
|
|
273
|
+
- name: Dispatch @argus mention
|
|
274
|
+
# issue_comment runs never check out the PR head; the workflow
|
|
275
|
+
# checks out the base ref and `mention` dispatches whitelisted
|
|
276
|
+
# commands (review/record/persist/help). Results still land in the
|
|
277
|
+
# report dir for the sticky post step below.
|
|
278
|
+
if: github.event_name == 'issue_comment'
|
|
279
|
+
id: mention
|
|
280
|
+
shell: bash
|
|
281
|
+
working-directory: ${{ inputs.working-directory || '.' }}
|
|
282
|
+
env:
|
|
283
|
+
ARGUS_ACTION_PATH: ${{ github.action_path }}
|
|
284
|
+
OPENROUTER_API_KEY: ${{ inputs.openrouter-api-key }}
|
|
285
|
+
GITHUB_TOKEN: ${{ github.token }}
|
|
286
|
+
ARGUS_CLI_JSON: ${{ steps.bootstrap.outputs.cli-json }}
|
|
287
|
+
ARGUS_WORKING_DIRECTORY: ${{ steps.bootstrap.outputs.working-directory }}
|
|
288
|
+
ARGUS_REPORT_DIR: ${{ inputs.report-dir }}
|
|
289
|
+
ARGUS_BUDGET_USD: ${{ inputs.budget-usd }}
|
|
290
|
+
ARGUS_MAX_COMMENTS: ${{ inputs.max-comments }}
|
|
291
|
+
ARGUS_CODE_MODEL: ${{ inputs.code-model }}
|
|
292
|
+
ARGUS_REVIEW_PROFILES: ${{ inputs.review-profiles }}
|
|
293
|
+
ARGUS_DEBUG: '1'
|
|
294
|
+
ARGUS_REVIEWER_TRACE: >-
|
|
295
|
+
{"repo":"${{ github.repository }}",
|
|
296
|
+
"pr":"${{ github.event.issue.number }}",
|
|
297
|
+
"commit":"${{ github.sha }}",
|
|
298
|
+
"run_id":"${{ github.run_id }}",
|
|
299
|
+
"run_attempt":"${{ github.run_attempt }}",
|
|
300
|
+
"workflow":"${{ github.workflow }}"}
|
|
301
|
+
run: node "$ARGUS_ACTION_PATH/cli.mjs" mention --report-dir "$ARGUS_REPORT_DIR"
|
|
302
|
+
|
|
189
303
|
- name: Run selected Argus lanes
|
|
304
|
+
if: github.event_name != 'issue_comment'
|
|
190
305
|
id: verify
|
|
191
306
|
shell: bash
|
|
192
307
|
working-directory: ${{ inputs.working-directory || '.' }}
|
|
@@ -199,11 +314,20 @@ runs:
|
|
|
199
314
|
ARGUS_WORKING_DIRECTORY: ${{ steps.bootstrap.outputs.working-directory }}
|
|
200
315
|
ARGUS_REPORT_DIR: ${{ inputs.report-dir }}
|
|
201
316
|
ARGUS_VERIFY_FLOW: ${{ inputs.run != 'false' && '1' || '' }}
|
|
317
|
+
ARGUS_VERIFY_APP: ${{ inputs.app == 'true' && '1' || '' }}
|
|
318
|
+
ARGUS_VERIFY_A0: ${{ inputs.a0 == 'true' && '1' || '' }}
|
|
319
|
+
ARGUS_VERIFY_URL: ${{ inputs.app-url }}
|
|
320
|
+
ARGUS_VERIFY_TASK: ${{ inputs.app-task }}
|
|
321
|
+
ARGUS_VERIFY_EXPECT_TEXT: ${{ inputs.app-expect-text }}
|
|
322
|
+
ARGUS_VERIFY_EXPECT_URL: ${{ inputs.app-expect-url }}
|
|
323
|
+
ARGUS_VERIFY_EXPECT_SELECTOR: ${{ inputs.app-expect-selector }}
|
|
202
324
|
ARGUS_SANDBOX: ${{ inputs.sandbox == 'true' && '1' || '' }}
|
|
203
325
|
ARGUS_MAX_COMMENTS: ${{ inputs.max-comments }}
|
|
204
326
|
ARGUS_DEBUG: '1'
|
|
205
327
|
ARGUS_DIFF_BASE: ${{ inputs.diff-base }}
|
|
206
328
|
ARGUS_BUDGET_USD: ${{ inputs.budget-usd }}
|
|
329
|
+
ARGUS_CODE_MODEL: ${{ inputs.code-model }}
|
|
330
|
+
ARGUS_REVIEW_PROFILES: ${{ inputs.review-profiles }}
|
|
207
331
|
ARGUS_REVIEWER_TRACE: >-
|
|
208
332
|
{"repo":"${{ github.repository }}",
|
|
209
333
|
"pr":"${{ github.event.pull_request.number }}",
|
|
@@ -20,6 +20,12 @@ import { execFile } from 'node:child_process'
|
|
|
20
20
|
import { readFile } from 'node:fs/promises'
|
|
21
21
|
import { join } from 'node:path'
|
|
22
22
|
|
|
23
|
+
// The Ocellus verdict vocabulary has one action-side copy (KTD2), in the
|
|
24
|
+
// sticky-comment module; this lane reuses it rather than keeping a third.
|
|
25
|
+
import sticky from './sticky-comment.cjs'
|
|
26
|
+
|
|
27
|
+
const { STATUS_GLYPH, VERDICT_STATUS, VERDICT_LABEL } = sticky
|
|
28
|
+
|
|
23
29
|
const GH_API = 'https://api.github.com'
|
|
24
30
|
|
|
25
31
|
/** Verdicts that mean "ship it". Everything else blocks. */
|
|
@@ -85,12 +91,16 @@ export function classifyApprovalFailure(status, message) {
|
|
|
85
91
|
export function reviewBody(codeReview, runUrl, evidence, evidenceCheck) {
|
|
86
92
|
const lines = []
|
|
87
93
|
const verdict = String(codeReview?.verdict ?? 'unknown')
|
|
88
|
-
lines.push(
|
|
94
|
+
lines.push(
|
|
95
|
+
Object.hasOwn(VERDICT_LABEL, verdict)
|
|
96
|
+
? `**Argus: ${STATUS_GLYPH[VERDICT_STATUS[verdict]]} ${VERDICT_LABEL[verdict]}**`
|
|
97
|
+
: `**Argus:** verdict \`${oneLine(verdict).replace(/`/g, '')}\``,
|
|
98
|
+
)
|
|
89
99
|
if (codeReview?.summary) lines.push('', codeReview.summary)
|
|
90
100
|
if (evidence !== undefined && evidence !== '') {
|
|
91
101
|
lines.push('', `**Verified by:** \`${oneLine(evidence)}\``)
|
|
92
102
|
} else {
|
|
93
|
-
lines.push('', '**Verified by:** none
|
|
103
|
+
lines.push('', '**Verified by:** none. This approval cites no test command.')
|
|
94
104
|
}
|
|
95
105
|
if (evidenceCheck) {
|
|
96
106
|
// The command above is the caller's claim; this line is the receipt. A
|
|
@@ -113,7 +123,7 @@ export function reviewBody(codeReview, runUrl, evidence, evidenceCheck) {
|
|
|
113
123
|
)
|
|
114
124
|
for (const f of blocking.slice(0, 5)) {
|
|
115
125
|
const where = f.file ? ` \`${f.file}${f.line ? `:${f.line}` : ''}\`` : ''
|
|
116
|
-
lines.push(`- **${f.severity}**${where}
|
|
126
|
+
lines.push(`- **${f.severity}**${where}: ${oneLine(f.message)}`)
|
|
117
127
|
}
|
|
118
128
|
}
|
|
119
129
|
if (runUrl) lines.push('', `[argus-reviewer run](${runUrl})`)
|
package/action/bootstrap.mjs
CHANGED
|
@@ -13,8 +13,10 @@ import {
|
|
|
13
13
|
setActionOutput,
|
|
14
14
|
validateBrowser,
|
|
15
15
|
validateBudget,
|
|
16
|
+
validateCodeModel,
|
|
16
17
|
validateMaxComments,
|
|
17
18
|
validatePathInput,
|
|
19
|
+
validateReviewProfiles,
|
|
18
20
|
validateVersion,
|
|
19
21
|
} from './runtime.mjs'
|
|
20
22
|
|
|
@@ -29,6 +31,8 @@ const reportDir = validatePathInput(
|
|
|
29
31
|
validateBrowser(process.env.ARGUS_BROWSER || 'chromium')
|
|
30
32
|
validateBudget(process.env.ARGUS_BUDGET_USD)
|
|
31
33
|
validateMaxComments(process.env.ARGUS_MAX_COMMENTS)
|
|
34
|
+
validateCodeModel(process.env.ARGUS_CODE_MODEL)
|
|
35
|
+
validateReviewProfiles(process.env.ARGUS_REVIEW_PROFILES)
|
|
32
36
|
|
|
33
37
|
async function stageConfig() {
|
|
34
38
|
if (configInput === '' || configInput === 'argus-reviewer.config.ts') return
|
package/action/emit-review.mjs
CHANGED
|
@@ -219,6 +219,22 @@ export async function emitApprovalReview(env, deps = {}) {
|
|
|
219
219
|
return { ok: false, reviewEvent: 'none', reviewState: 'no-report', message }
|
|
220
220
|
}
|
|
221
221
|
|
|
222
|
+
// The report must be this run's own before it can drive any review event.
|
|
223
|
+
// Head sha alone is forgeable (a commit author knows it); GITHUB_RUN_ID is
|
|
224
|
+
// not knowable when a commit or planted file is authored, so a report that
|
|
225
|
+
// cannot present it is residue or plant. Local runs have no run id — the
|
|
226
|
+
// gate is off by design there, same as the sticky post step.
|
|
227
|
+
const expectedNonce = (env.GITHUB_RUN_ID ?? '').trim()
|
|
228
|
+
if (expectedNonce !== '' && codeReview?.runNonce !== expectedNonce) {
|
|
229
|
+
const message =
|
|
230
|
+
'argus-reviewer: code-review.json does not carry this workflow run\'s id — ' +
|
|
231
|
+
'it is stale or was not produced by this run. No review was submitted.'
|
|
232
|
+
fail(message)
|
|
233
|
+
setActionOutput('review-event', 'none')
|
|
234
|
+
setActionOutput('review-state', 'stale-report')
|
|
235
|
+
return { ok: false, reviewEvent: 'none', reviewState: 'stale-report', message }
|
|
236
|
+
}
|
|
237
|
+
|
|
222
238
|
const event = reviewEventFor(codeReview.verdict)
|
|
223
239
|
|
|
224
240
|
// The run reviewed the code at the head SHA its event carried. If the head has
|
package/action/runtime.mjs
CHANGED
|
@@ -93,6 +93,38 @@ export function validateMaxComments(raw) {
|
|
|
93
93
|
return Number(raw)
|
|
94
94
|
}
|
|
95
95
|
|
|
96
|
+
const KNOWN_REVIEW_PROFILES = new Set(['security', 'perf', 'debloat'])
|
|
97
|
+
|
|
98
|
+
export function validateReviewProfiles(raw) {
|
|
99
|
+
if (raw === undefined || raw === '') return undefined
|
|
100
|
+
const names = String(raw)
|
|
101
|
+
.split(',')
|
|
102
|
+
.map((s) => s.trim())
|
|
103
|
+
.filter((s) => s !== '')
|
|
104
|
+
// Env-only value — no shell — but reject unknown names so a typo fails
|
|
105
|
+
// here instead of silently dropping the intended lens at config load.
|
|
106
|
+
for (const name of names) {
|
|
107
|
+
if (!KNOWN_REVIEW_PROFILES.has(name)) {
|
|
108
|
+
throw new Error(
|
|
109
|
+
`review-profiles must be a comma-separated list of: ${[...KNOWN_REVIEW_PROFILES].join(', ')} — got '${name}'`,
|
|
110
|
+
)
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
return names.join(',')
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
export function validateCodeModel(raw) {
|
|
117
|
+
if (raw === undefined || raw === '') return undefined
|
|
118
|
+
const value = String(raw).trim()
|
|
119
|
+
// OpenRouter slugs are `vendor/model` with an optional `:variant`
|
|
120
|
+
// (`deepseek/deepseek-r1:free`). Env-only value — no shell — but keep it
|
|
121
|
+
// to slug-shaped input so typos surface here instead of at the API.
|
|
122
|
+
if (!/^[A-Za-z0-9][A-Za-z0-9._:/-]{0,199}$/.test(value)) {
|
|
123
|
+
throw new Error('code-model must be an OpenRouter model slug (e.g. openai/gpt-5-mini)')
|
|
124
|
+
}
|
|
125
|
+
return value
|
|
126
|
+
}
|
|
127
|
+
|
|
96
128
|
export function resolveWorkingDirectory(workspace, input) {
|
|
97
129
|
const root = realpathSync(resolve(workspace))
|
|
98
130
|
const requested = String(input ?? '').trim()
|