argus-reviewer-e2e 0.4.1 → 0.4.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +32 -5
- package/action/action.yml +3 -3
- package/action/runtime.mjs +2 -2
- package/dist/cli.d.ts +18 -0
- package/dist/cli.js +259 -237
- package/dist/config.d.ts +60 -2
- package/dist/config.js +85 -2
- package/dist/detect.js +2 -2
- package/dist/engine/loop.d.ts +8 -1
- package/dist/onboarding/pr-content.d.ts +12 -0
- package/dist/onboarding/pr-content.js +56 -0
- package/dist/onboarding/pr.d.ts +23 -0
- package/dist/onboarding/pr.js +196 -0
- package/dist/onboarding/scaffold.d.ts +37 -0
- package/dist/onboarding/scaffold.js +174 -0
- package/dist/pipeline/budget.d.ts +13 -0
- package/dist/pipeline/budget.js +34 -0
- package/dist/review/chunks.d.ts +21 -0
- package/dist/review/chunks.js +96 -0
- package/dist/vision/openrouter.d.ts +37 -1
- package/dist/vision/openrouter.js +111 -5
- package/package.json +2 -1
package/README.md
CHANGED
|
@@ -16,7 +16,24 @@
|
|
|
16
16
|
|
|
17
17
|
It is a GitHub Action and a CLI. It runs on your infrastructure with your own OpenRouter key: no hosted service, no telemetry, no per-seat pricing. MIT licensed.
|
|
18
18
|
|
|
19
|
-
##
|
|
19
|
+
## Get started
|
|
20
|
+
|
|
21
|
+
From a checkout of your GitHub repository, with `git` and the GitHub CLI (`gh auth login`) set up:
|
|
22
|
+
|
|
23
|
+
```bash
|
|
24
|
+
npx argus-reviewer init --pr
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
This opens a pull request that adds the Argus workflows, a config and a smoke test. Nothing runs until you merge it. Then:
|
|
28
|
+
|
|
29
|
+
1. Add `OPENROUTER_API_KEY` as a repository secret (the PR links to the exact page, or run `gh secret set OPENROUTER_API_KEY --repo owner/name`).
|
|
30
|
+
2. Review and merge the PR. Every pull request after that gets a review.
|
|
31
|
+
|
|
32
|
+
`init --pr` uses your own `git` and `gh`, so your credentials do the pushing. Optional flags: `--repo owner/name` (must match the checkout's `origin`) and `--branch <name>` (default `argus/onboarding`). It never overwrites existing files, reports the existing PR instead of opening a second one, and never reads your OpenRouter key. Details, defaults, what is sent to the model provider, and troubleshooting: [`docs/onboarding.md`](docs/onboarding.md).
|
|
33
|
+
|
|
34
|
+
### Manual setup
|
|
35
|
+
|
|
36
|
+
To write the files into your working tree instead of opening a PR:
|
|
20
37
|
|
|
21
38
|
```bash
|
|
22
39
|
npm i -D argus-reviewer-e2e # the package; the command is argus-reviewer
|
|
@@ -25,7 +42,11 @@ npx argus-reviewer init # config, a smoke test, and the PR workflow
|
|
|
25
42
|
|
|
26
43
|
<img src="docs/assets/demo/init.gif" width="1100" alt="Terminal: argus-reviewer init writes the config, a smoke test and two workflow files, then checks the environment. The OpenRouter key is reported as not set, Playwright chromium is found, and the default lane is code review." />
|
|
27
44
|
|
|
28
|
-
`init` checks your environment and tells you what is missing. Add `OPENROUTER_API_KEY` to your shell and to the repository secrets, and every pull request gets a review.
|
|
45
|
+
`init` checks your environment and tells you what is missing. Add `OPENROUTER_API_KEY` to your shell and to the repository secrets, commit the files, and every pull request gets a review.
|
|
46
|
+
|
|
47
|
+
A GitHub App that opens the onboarding PR for you on install is planned and not available yet; see [`docs/onboarding.md`](docs/onboarding.md).
|
|
48
|
+
|
|
49
|
+
To have an App open that PR when you install it on a repository, register and host your own: [`docs/self-host-app.md`](docs/self-host-app.md).
|
|
29
50
|
|
|
30
51
|
Record a browser flow once, then replay it on every run:
|
|
31
52
|
|
|
@@ -94,7 +115,7 @@ Every lane, in the comment and in the terminal, reports one of six statuses:
|
|
|
94
115
|
|
|
95
116
|
## Cost
|
|
96
117
|
|
|
97
|
-
Every OpenRouter call is metered from the provider's per-call price and totaled in the comment. The review in the image above cost **$0.000739**. A flow replay that matches its cache costs $0. `budgetUsd` caps each run (default $1.00), and you choose the model for each job (`model`, `code_model`, `escalation_model`).
|
|
118
|
+
Every OpenRouter call is metered from the provider's per-call price and totaled in the comment. The review in the image above cost **$0.000739**. A flow replay that matches its cache costs $0. `budgetUsd` caps each run (default $1.00; raise it, or set `0` to run uncapped, which logs a warning), and you choose the model for each job (`model`, `code_model`, `escalation_model`).
|
|
98
119
|
|
|
99
120
|
The action tags every call with `ARGUS_REVIEWER_TRACE` (repository, PR, commit, run), so spend can be attributed per review. See [`docs/quickstart.md`](docs/quickstart.md) for the `openrouter` config block.
|
|
100
121
|
|
|
@@ -111,9 +132,9 @@ import { defineConfig } from 'argus-reviewer-e2e'
|
|
|
111
132
|
|
|
112
133
|
export default defineConfig({
|
|
113
134
|
model: 'google/gemini-2.5-flash-lite', // vision: grounding and actions
|
|
114
|
-
code_model: 'deepseek/deepseek-v4
|
|
135
|
+
code_model: 'deepseek/deepseek-v4-flash', // diff review
|
|
115
136
|
escalation_model: 'anthropic/claude-sonnet-4', // risky or complex findings
|
|
116
|
-
budgetUsd: 1.0,
|
|
137
|
+
budgetUsd: 1.0, // per-run cap in USD; default 1, 0 = unlimited
|
|
117
138
|
target: { url: 'https://your-app.example.com' },
|
|
118
139
|
testsDir: 'e2e',
|
|
119
140
|
reportRetention: 20, // archived manifests to keep
|
|
@@ -122,6 +143,12 @@ export default defineConfig({
|
|
|
122
143
|
|
|
123
144
|
Code review skips generated, fixture and vendored paths by default (`dist/**`, `fixtures/**`, `tests/goldens/**`, lockfiles, `*.generated.*`, `assets/brand/export/**`). Set `review: { exclude: [...] }` to replace that list (`[]` excludes nothing). The sticky comment's Diagnostics fold says how many files were left out.
|
|
124
145
|
|
|
146
|
+
Large PRs are reviewed in chunks (about 6k tokens of diff each, grouped by directory; a single oversized file is split at hunk boundaries) and the findings are merged. The review summary says how many chunks and files were reviewed. When `codeReviewBudgetUsd` cannot cover the next chunk, the run stops before spending it and the summary lists how many files went unreviewed.
|
|
147
|
+
|
|
148
|
+
Batch mode: `review: { mode: 'batch' }` (or `--mode batch`, or `ARGUS_REVIEW_MODE=batch`; default `realtime`) sends all chunks through OpenRouter's async Batch API instead of one call each. It is slower (minutes; a probe took about six) and is polled until `review.batchTimeoutMs` (default 480000, kept inside the 15-minute job timeout). On failure, timeout, or a single errored request, Argus falls back to realtime for the affected chunks. Cost is metered from the batch usage, and `code-review.json` records `batch.used` / `batch.fellBack`.
|
|
149
|
+
|
|
150
|
+
Models and timeouts: realtime review uses `code_model` (default `deepseek/deepseek-v4-flash`: cheap but noisier, so the validate step and severity gating stay on). Batch uses `review.batchModel` (`--batch-model`, `ARGUS_BATCH_MODEL`; default `deepseek/deepseek-v4.1-flash:batch`, or `<code_model>:batch` when that model is known to have a batch endpoint). Batch is the recommended mode for large PRs. `review.requestTimeoutMs` (`ARGUS_REQUEST_TIMEOUT_MS`; default 120000, max 900000) is the per-request timeout; raise it for reasoning models such as `deepseek/deepseek-v4.1-flash`. To keep the previous review model, set `code_model: 'deepseek/deepseek-v4.1-flash'` with `review: { requestTimeoutMs: 600000 }`.
|
|
151
|
+
|
|
125
152
|
Full shape: [`src/config.ts`](src/config.ts). Setup walkthrough: [`docs/quickstart.md`](docs/quickstart.md).
|
|
126
153
|
|
|
127
154
|
## Contributing
|
package/action/action.yml
CHANGED
|
@@ -24,7 +24,7 @@ inputs:
|
|
|
24
24
|
description: Working directory for the consumer's project
|
|
25
25
|
required: false
|
|
26
26
|
budget-usd:
|
|
27
|
-
description:
|
|
27
|
+
description: Per-run spend cap in USD. Default 1 (built in). Set 0 to run uncapped (logs a warning).
|
|
28
28
|
required: false
|
|
29
29
|
index:
|
|
30
30
|
description: Run `argus-reviewer index` before the test run to enable diff-aware cache invalidation
|
|
@@ -47,7 +47,7 @@ inputs:
|
|
|
47
47
|
default: argus-reviewer-report
|
|
48
48
|
required: false
|
|
49
49
|
sandbox:
|
|
50
|
-
description: Enable the B.2 probe lane: authored test probes for unexercised findings run in a hardened Docker container (no network, no secrets, read-only fs). Enable-only: 'false' does not override a config-enabled lane. Requires Docker on the runner; forced off on pull_request_target.
|
|
50
|
+
description: "Enable the B.2 probe lane: authored test probes for unexercised findings run in a hardened Docker container (no network, no secrets, read-only fs). Enable-only: 'false' does not override a config-enabled lane. Requires Docker on the runner; forced off on pull_request_target."
|
|
51
51
|
default: 'false'
|
|
52
52
|
required: false
|
|
53
53
|
max-comments:
|
|
@@ -59,7 +59,7 @@ inputs:
|
|
|
59
59
|
default: ''
|
|
60
60
|
required: false
|
|
61
61
|
review-profiles:
|
|
62
|
-
description: Optional comma-separated review lenses appended to the review prompt: security, perf, debloat (e.g. 'security,perf'). Wins over the checkout's config review.profiles.
|
|
62
|
+
description: "Optional comma-separated review lenses appended to the review prompt: security, perf, debloat (e.g. 'security,perf'). Wins over the checkout's config review.profiles."
|
|
63
63
|
default: ''
|
|
64
64
|
required: false
|
|
65
65
|
install-consumer-dependencies:
|
package/action/runtime.mjs
CHANGED
|
@@ -79,8 +79,8 @@ export function validateBrowser(raw) {
|
|
|
79
79
|
export function validateBudget(raw, name = 'budget-usd') {
|
|
80
80
|
if (raw === undefined || raw === '') return undefined
|
|
81
81
|
const value = Number(raw)
|
|
82
|
-
if (!Number.isFinite(value) || value
|
|
83
|
-
throw new Error(`${name} must be a
|
|
82
|
+
if (!Number.isFinite(value) || value < 0) {
|
|
83
|
+
throw new Error(`${name} must be a non-negative number (0 = unlimited)`)
|
|
84
84
|
}
|
|
85
85
|
return value
|
|
86
86
|
}
|
package/dist/cli.d.ts
CHANGED
|
@@ -74,6 +74,16 @@ export interface DroppedFinding {
|
|
|
74
74
|
message: string;
|
|
75
75
|
reason: 'outside-diff' | 'revert-nit';
|
|
76
76
|
}
|
|
77
|
+
export interface ReviewBatch {
|
|
78
|
+
/** True when the Batch API produced the chunk reviews. */
|
|
79
|
+
used: boolean;
|
|
80
|
+
/** Chunks submitted. */
|
|
81
|
+
chunks: number;
|
|
82
|
+
/** Batch chunks re-run realtime because their request errored. */
|
|
83
|
+
retriedRealtime?: number;
|
|
84
|
+
/** Why the whole batch fell back to realtime. */
|
|
85
|
+
fellBack?: string;
|
|
86
|
+
}
|
|
77
87
|
export interface ReviewScope {
|
|
78
88
|
/** Changed files in the PR with a patch. */
|
|
79
89
|
totalFiles: number;
|
|
@@ -81,6 +91,12 @@ export interface ReviewScope {
|
|
|
81
91
|
reviewedFiles: number;
|
|
82
92
|
/** Files kept out by `review.exclude`. */
|
|
83
93
|
excludedFiles: number;
|
|
94
|
+
/** Model calls the diff was split into (1 for a PR that fits one call). */
|
|
95
|
+
chunksTotal?: number;
|
|
96
|
+
/** Chunks that were actually reviewed (fewer than total when the budget stopped the run). */
|
|
97
|
+
chunksReviewed?: number;
|
|
98
|
+
/** Reviewed files with no chunk reviewed (budget stop); 0 on a full review. */
|
|
99
|
+
unreviewedFiles?: number;
|
|
84
100
|
/** Up to 5 excluded paths, for the Diagnostics line. */
|
|
85
101
|
excludedSample: string[];
|
|
86
102
|
}
|
|
@@ -124,6 +140,8 @@ interface CodeReviewReport {
|
|
|
124
140
|
modelVerdict?: 'pass' | 'needs_changes' | 'approve';
|
|
125
141
|
/** How much of the PR the review covered, and what was left out. */
|
|
126
142
|
scope?: ReviewScope;
|
|
143
|
+
/** Present when `review.mode` is batch: whether the batch served the review. */
|
|
144
|
+
batch?: ReviewBatch;
|
|
127
145
|
/** Findings dropped by deterministic validation, with reasons. */
|
|
128
146
|
validation?: ValidationAudit;
|
|
129
147
|
/** Test-file findings capped at nit (bug/risk with no non-test citation). */
|