argus-reviewer-e2e 0.4.0 → 0.4.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +32 -5
- package/action/action.yml +3 -3
- package/action/runtime.mjs +2 -2
- package/dist/cli.d.ts +18 -0
- package/dist/cli.js +259 -237
- package/dist/config.d.ts +60 -2
- package/dist/config.js +85 -2
- package/dist/detect.js +2 -2
- package/dist/engine/loop.d.ts +8 -1
- package/dist/onboarding/pr-content.d.ts +12 -0
- package/dist/onboarding/pr-content.js +56 -0
- package/dist/onboarding/pr.d.ts +23 -0
- package/dist/onboarding/pr.js +196 -0
- package/dist/onboarding/scaffold.d.ts +37 -0
- package/dist/onboarding/scaffold.js +174 -0
- package/dist/pipeline/budget.d.ts +13 -0
- package/dist/pipeline/budget.js +34 -0
- package/dist/review/chunks.d.ts +21 -0
- package/dist/review/chunks.js +96 -0
- package/dist/vision/openrouter.d.ts +37 -1
- package/dist/vision/openrouter.js +111 -5
- package/package.json +2 -1
|
@@ -0,0 +1,174 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Pure scaffold generator shared by `argus-reviewer init` and `init --pr`
|
|
3
|
+
* (and, later, the App's onboarding Worker, so the surfaces cannot drift).
|
|
4
|
+
* No I/O, no config execution, no environment reads. Workflow templates keep
|
|
5
|
+
* `persist-credentials: false`, a pinned action SHA, least-privilege
|
|
6
|
+
* `permissions`, and never `pull_request_target`.
|
|
7
|
+
*/
|
|
8
|
+
/**
|
|
9
|
+
* The one place the user-facing action pin lives. Bump it here (and the golden
|
|
10
|
+
* fixtures) on release; see RELEASING.md. v0.4.0 and v0.4.1 ship an action.yml
|
|
11
|
+
* GitHub cannot parse, so never pin to them.
|
|
12
|
+
*/
|
|
13
|
+
export const ACTION_PIN_SHA = 'dbf4b7f669507402d527cb0e350285ca56129448';
|
|
14
|
+
export const ACTION_PIN_TAG = 'v0.4.2';
|
|
15
|
+
const ACTION_PIN = `${ACTION_PIN_SHA} # ${ACTION_PIN_TAG}`;
|
|
16
|
+
export function initConfig(a0Host) {
|
|
17
|
+
// R19 — a detected Agent Zero host earns a labeled suggestion, never an
|
|
18
|
+
// enabled lane: `verify --a0` is explicit opt-in per run, and completed
|
|
19
|
+
// delegations cap at inconclusive (self-reported evidence).
|
|
20
|
+
const a0Block = a0Host !== undefined
|
|
21
|
+
? `
|
|
22
|
+
// Optional: Agent Zero detected at ${a0Host}. Nothing below runs unless
|
|
23
|
+
// you ask for it — both stays commented until you opt in deliberately.
|
|
24
|
+
// a0: { url: ${JSON.stringify(a0Host)} }, // enables \`verify --a0\` (self-reported, unmetered)
|
|
25
|
+
// heal: 'a0', // escalates a failed heal to the A0 host
|
|
26
|
+
`
|
|
27
|
+
: '';
|
|
28
|
+
return `import { defineConfig } from 'argus-reviewer-e2e'
|
|
29
|
+
|
|
30
|
+
export default defineConfig({
|
|
31
|
+
// The app under test. command boots it (omit if it is already running);
|
|
32
|
+
// argus-reviewer polls url until it responds before running tests.
|
|
33
|
+
target: {
|
|
34
|
+
command: 'npm run dev',
|
|
35
|
+
url: 'http://localhost:3000',
|
|
36
|
+
readyTimeoutMs: 30_000,
|
|
37
|
+
},
|
|
38
|
+
// Hard per-run cap on vision-model spend (USD). Steps replayed from the
|
|
39
|
+
// fingerprint cache cost $0 regardless of this cap.
|
|
40
|
+
budgetUsd: 1,
|
|
41
|
+
testsDir: 'tests/argus',
|
|
42
|
+
// Exploratory lane: after the test loop, a bounded agent pass probes the
|
|
43
|
+
// app itself — same-origin navigation, clicks, invalid input — while taps
|
|
44
|
+
// capture console errors, page errors, and failed requests. Findings
|
|
45
|
+
// render as 'observed' — evidence only, never verdict-changing.
|
|
46
|
+
// maxSteps caps acts per run; budgetUsd caps explore model spend (falls
|
|
47
|
+
// back to budgetUsd). Point it at disposable targets only — clicks and
|
|
48
|
+
// form submits have real side effects.
|
|
49
|
+
// explore: { enabled: true, maxSteps: 20, budgetUsd: 0.25 },${a0Block}
|
|
50
|
+
})
|
|
51
|
+
`;
|
|
52
|
+
}
|
|
53
|
+
export const INIT_TEST = `test('home renders', async (td) => {
|
|
54
|
+
const ok = await td.assert('the page rendered without obvious errors')
|
|
55
|
+
if (!ok) throw new Error('home did not render')
|
|
56
|
+
})
|
|
57
|
+
`;
|
|
58
|
+
export const INIT_WORKFLOW = `name: argus-reviewer
|
|
59
|
+
|
|
60
|
+
on:
|
|
61
|
+
pull_request:
|
|
62
|
+
# 'labeled' lets a maintainer re-trigger with the argus-probe label when
|
|
63
|
+
# sandbox probes are enabled for fork PRs.
|
|
64
|
+
types: [opened, synchronize, reopened, labeled]
|
|
65
|
+
|
|
66
|
+
jobs:
|
|
67
|
+
argus:
|
|
68
|
+
runs-on: ubuntu-latest
|
|
69
|
+
# 'labeled' fires on EVERY label — only argus-probe is the fork-gate
|
|
70
|
+
# signal worth a full review run.
|
|
71
|
+
if: github.event.action != 'labeled' || github.event.label.name == 'argus-probe'
|
|
72
|
+
permissions:
|
|
73
|
+
contents: read
|
|
74
|
+
issues: write
|
|
75
|
+
pull-requests: write
|
|
76
|
+
checks: write
|
|
77
|
+
statuses: write
|
|
78
|
+
steps:
|
|
79
|
+
# persist-credentials: false keeps the GITHUB_TOKEN out of .git/config —
|
|
80
|
+
# the probe sandbox masks .git regardless, but don't store it at all.
|
|
81
|
+
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
|
82
|
+
with:
|
|
83
|
+
persist-credentials: false
|
|
84
|
+
ref: \${{ github.event.pull_request.head.sha || github.sha }}
|
|
85
|
+
# Optional verdict-as-review: let Argus submit APPROVE / REQUEST_CHANGES
|
|
86
|
+
# so require_approving_reviews counts it. GITHUB_TOKEN cannot approve, so
|
|
87
|
+
# create + install your own GitHub App (docs/github-app.md), set the
|
|
88
|
+
# ARGUS_APP_ID variable and ARGUS_APP_PRIVATE_KEY secret, then uncomment:
|
|
89
|
+
# - uses: actions/create-github-app-token@fee1f7d63c2ff003460e3d139729b119787bc349 # v2
|
|
90
|
+
# id: argus-app
|
|
91
|
+
# with:
|
|
92
|
+
# app-id: \${{ vars.ARGUS_APP_ID }}
|
|
93
|
+
# private-key: \${{ secrets.ARGUS_APP_PRIVATE_KEY }}
|
|
94
|
+
# and pass approval-token plus its evidence inputs to the action below:
|
|
95
|
+
# approval-token: \${{ steps.argus-app.outputs.token }}
|
|
96
|
+
# approval-evidence: 'npm test' # command the approval stands on
|
|
97
|
+
# approval-check: 'test' # check-run name, green on head SHA
|
|
98
|
+
- uses: duketopceo/Argus/action@@@ACTION_PIN@@
|
|
99
|
+
with:
|
|
100
|
+
openrouter-api-key: \${{ secrets.OPENROUTER_API_KEY }}
|
|
101
|
+
`.replace('@@ACTION_PIN@@', ACTION_PIN);
|
|
102
|
+
export const INIT_MENTION_WORKFLOW = `name: argus-mention
|
|
103
|
+
|
|
104
|
+
# @argus mention commands on PR comments — '@argus review', '@argus
|
|
105
|
+
# record "<flow>"', '@argus persist', '@argus help'. issue_comment is
|
|
106
|
+
# strictly more privileged than pull_request (secrets + write token are
|
|
107
|
+
# present), so the checkout below deliberately resolves the BASE ref —
|
|
108
|
+
# never the PR head. Argus reviews the head diff over the API.
|
|
109
|
+
on:
|
|
110
|
+
issue_comment:
|
|
111
|
+
types: [created]
|
|
112
|
+
|
|
113
|
+
jobs:
|
|
114
|
+
argus-mention:
|
|
115
|
+
runs-on: ubuntu-latest
|
|
116
|
+
if: github.event.issue.pull_request && startsWith(github.event.comment.body, '@argus')
|
|
117
|
+
permissions:
|
|
118
|
+
# contents: write — '@argus persist' commits reproduced probes to an
|
|
119
|
+
# argus/ branch via the git/refs + contents APIs and opens a PR.
|
|
120
|
+
contents: write
|
|
121
|
+
issues: write
|
|
122
|
+
pull-requests: write
|
|
123
|
+
checks: write
|
|
124
|
+
statuses: write
|
|
125
|
+
steps:
|
|
126
|
+
# No 'ref' — the default checkout resolves the base branch. persist
|
|
127
|
+
# writes via the API, so checkout credentials stay disabled.
|
|
128
|
+
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
|
|
129
|
+
with:
|
|
130
|
+
persist-credentials: false
|
|
131
|
+
# Record commands need the app's dependencies to boot its target.
|
|
132
|
+
# Uncomment if you use '@argus record':
|
|
133
|
+
# - run: npm ci
|
|
134
|
+
- uses: duketopceo/Argus/action@@@ACTION_PIN@@
|
|
135
|
+
with:
|
|
136
|
+
openrouter-api-key: \${{ secrets.OPENROUTER_API_KEY }}
|
|
137
|
+
# '@argus record' uploads the generated test + flow cache as an
|
|
138
|
+
# artifact — committing to a PR branch is intentionally not done.
|
|
139
|
+
- uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
|
140
|
+
if: contains(github.event.comment.body, 'record')
|
|
141
|
+
with:
|
|
142
|
+
name: argus-recorded-flow
|
|
143
|
+
path: |
|
|
144
|
+
tests/argus/
|
|
145
|
+
.argus-reviewer-cache/
|
|
146
|
+
if-no-files-found: ignore
|
|
147
|
+
`.replace('@@ACTION_PIN@@', ACTION_PIN);
|
|
148
|
+
export const CONFIG_PATH = 'argus-reviewer.config.ts';
|
|
149
|
+
/** The files `init` writes, in write order. */
|
|
150
|
+
export function renderScaffold(opts) {
|
|
151
|
+
const files = [
|
|
152
|
+
{ path: 'tests/argus/smoke.test.ts', content: INIT_TEST },
|
|
153
|
+
{ path: '.github/workflows/argus-reviewer.yml', content: INIT_WORKFLOW },
|
|
154
|
+
{ path: '.github/workflows/argus-mention.yml', content: INIT_MENTION_WORKFLOW },
|
|
155
|
+
];
|
|
156
|
+
if (opts.includeConfig)
|
|
157
|
+
files.unshift({ path: CONFIG_PATH, content: initConfig(opts.a0Host) });
|
|
158
|
+
return files;
|
|
159
|
+
}
|
|
160
|
+
/**
|
|
161
|
+
* "What runs and what it costs": what is sent to the provider, the default
|
|
162
|
+
* budget, and the stop path. Kept verbatim (DESIGN 7.8). `budgetUsd` comes
|
|
163
|
+
* from the caller's resolved defaults so it cannot go stale here.
|
|
164
|
+
*/
|
|
165
|
+
export function scaffoldChecklist(budgetUsd) {
|
|
166
|
+
return [
|
|
167
|
+
'What runs and what it costs:',
|
|
168
|
+
' sent to provider PR diffs, page screenshots/DOM snapshots, and',
|
|
169
|
+
' review prompts — via your OpenRouter key (BYOK)',
|
|
170
|
+
` default budget $${budgetUsd}/run cap (budgetUsd); cached replay costs $0`,
|
|
171
|
+
' how to stop Ctrl+C locally; in CI remove the workflow file',
|
|
172
|
+
' or delete the OPENROUTER_API_KEY secret',
|
|
173
|
+
];
|
|
174
|
+
}
|
|
@@ -10,3 +10,16 @@ export declare function overLimit(spentUsd: number, limitUsd: number): boolean;
|
|
|
10
10
|
export declare function addProviderCalls(budget: BudgetSummary, calls: CallCost[] | undefined): BudgetSummary;
|
|
11
11
|
export declare function addA0Task(budget: BudgetSummary, elapsedMs: number, metered: boolean): BudgetSummary;
|
|
12
12
|
export declare function budgetCanSpend(budget: BudgetSummary, nextCostUsd: number): boolean;
|
|
13
|
+
export declare function estimateRequestCostUsd(req: {
|
|
14
|
+
messages: unknown;
|
|
15
|
+
schema?: unknown;
|
|
16
|
+
}): number;
|
|
17
|
+
/**
|
|
18
|
+
* How many leading requests of a batch fit in the remaining budget. A batch
|
|
19
|
+
* cannot be cancelled once submitted, so the guard sizes it up front.
|
|
20
|
+
* `limitUsd` undefined = unlimited.
|
|
21
|
+
*/
|
|
22
|
+
export declare function affordableBatchPrefix(requests: ReadonlyArray<{
|
|
23
|
+
messages: unknown;
|
|
24
|
+
schema?: unknown;
|
|
25
|
+
}>, limitUsd: number | undefined, spentUsd: number): number;
|
package/dist/pipeline/budget.js
CHANGED
|
@@ -46,3 +46,37 @@ export function budgetCanSpend(budget, nextCostUsd) {
|
|
|
46
46
|
(budget.limitUsd === undefined ||
|
|
47
47
|
budget.spentUsd + nextCostUsd <= budget.limitUsd + USD_EPSILON));
|
|
48
48
|
}
|
|
49
|
+
/**
|
|
50
|
+
* Conservative per-request cost ceiling (USD) used ONLY where a paid call
|
|
51
|
+
* cannot be stopped mid-flight (batch submission). Argus has no live price
|
|
52
|
+
* table, so this prices tokens at a deliberately high blended rate (well above
|
|
53
|
+
* the default review models) and reserves a fixed output allowance. Real
|
|
54
|
+
* spend is metered from provider usage; this only decides whether to submit.
|
|
55
|
+
*/
|
|
56
|
+
const EST_INPUT_USD_PER_TOKEN = 3 / 1_000_000;
|
|
57
|
+
const EST_OUTPUT_USD_PER_TOKEN = 12 / 1_000_000;
|
|
58
|
+
const EST_OUTPUT_TOKENS = 4_000;
|
|
59
|
+
const EST_CHARS_PER_TOKEN = 3;
|
|
60
|
+
export function estimateRequestCostUsd(req) {
|
|
61
|
+
const chars = JSON.stringify(req.messages ?? '').length + JSON.stringify(req.schema ?? '').length;
|
|
62
|
+
const inputTokens = Math.ceil(chars / EST_CHARS_PER_TOKEN);
|
|
63
|
+
return inputTokens * EST_INPUT_USD_PER_TOKEN + EST_OUTPUT_TOKENS * EST_OUTPUT_USD_PER_TOKEN;
|
|
64
|
+
}
|
|
65
|
+
/**
|
|
66
|
+
* How many leading requests of a batch fit in the remaining budget. A batch
|
|
67
|
+
* cannot be cancelled once submitted, so the guard sizes it up front.
|
|
68
|
+
* `limitUsd` undefined = unlimited.
|
|
69
|
+
*/
|
|
70
|
+
export function affordableBatchPrefix(requests, limitUsd, spentUsd) {
|
|
71
|
+
if (limitUsd === undefined)
|
|
72
|
+
return requests.length;
|
|
73
|
+
let projected = spentUsd;
|
|
74
|
+
let n = 0;
|
|
75
|
+
for (const r of requests) {
|
|
76
|
+
projected += estimateRequestCostUsd(r);
|
|
77
|
+
if (projected > limitUsd + USD_EPSILON)
|
|
78
|
+
break;
|
|
79
|
+
n++;
|
|
80
|
+
}
|
|
81
|
+
return n;
|
|
82
|
+
}
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Chunk planning for large PRs.
|
|
3
|
+
*
|
|
4
|
+
* A diff over the per-call token target is reviewed in several model calls
|
|
5
|
+
* instead of one oversized (or truncated) prompt. Files are grouped by
|
|
6
|
+
* directory so a chunk reads as one area of the change, and a single patch
|
|
7
|
+
* larger than the target is split at hunk boundaries. Every chunk records
|
|
8
|
+
* which files it carries so a partial review can say what it did not cover.
|
|
9
|
+
*/
|
|
10
|
+
export declare const CHUNK_TOKEN_TARGET = 6000;
|
|
11
|
+
export interface ChunkFile {
|
|
12
|
+
filename: string;
|
|
13
|
+
patch?: string;
|
|
14
|
+
}
|
|
15
|
+
export interface PlannedChunk {
|
|
16
|
+
/** Prompt text for this chunk. */
|
|
17
|
+
text: string;
|
|
18
|
+
/** Files with content in this chunk (a split file appears in each part). */
|
|
19
|
+
files: string[];
|
|
20
|
+
}
|
|
21
|
+
export declare function planChunks(files: ChunkFile[], contexts?: Record<string, string>, targetTokens?: number): PlannedChunk[];
|
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Chunk planning for large PRs.
|
|
3
|
+
*
|
|
4
|
+
* A diff over the per-call token target is reviewed in several model calls
|
|
5
|
+
* instead of one oversized (or truncated) prompt. Files are grouped by
|
|
6
|
+
* directory so a chunk reads as one area of the change, and a single patch
|
|
7
|
+
* larger than the target is split at hunk boundaries. Every chunk records
|
|
8
|
+
* which files it carries so a partial review can say what it did not cover.
|
|
9
|
+
*/
|
|
10
|
+
export const CHUNK_TOKEN_TARGET = 6000;
|
|
11
|
+
const CHUNK_FILE_OVERHEAD = 100;
|
|
12
|
+
const tokensOf = (s) => Math.ceil(s.length / 4);
|
|
13
|
+
function dirOf(filename) {
|
|
14
|
+
const i = filename.lastIndexOf('/');
|
|
15
|
+
return i < 0 ? '' : filename.slice(0, i);
|
|
16
|
+
}
|
|
17
|
+
/** Stable group-by-directory; groups keep first-appearance order. */
|
|
18
|
+
function groupByDir(files) {
|
|
19
|
+
const groups = new Map();
|
|
20
|
+
for (const f of files) {
|
|
21
|
+
const d = dirOf(f.filename);
|
|
22
|
+
const g = groups.get(d);
|
|
23
|
+
if (g === undefined)
|
|
24
|
+
groups.set(d, [f]);
|
|
25
|
+
else
|
|
26
|
+
g.push(f);
|
|
27
|
+
}
|
|
28
|
+
return [...groups.values()].flat();
|
|
29
|
+
}
|
|
30
|
+
/** Split a patch at `@@` hunk starts into parts that each fit `budgetTokens`. */
|
|
31
|
+
function splitPatch(patch, budgetTokens) {
|
|
32
|
+
const firstHunk = patch.search(/^@@/m);
|
|
33
|
+
if (firstHunk < 0)
|
|
34
|
+
return [patch];
|
|
35
|
+
const header = patch.slice(0, firstHunk);
|
|
36
|
+
const hunks = patch.slice(firstHunk).split(/^(?=@@)/m);
|
|
37
|
+
const parts = [];
|
|
38
|
+
let cur = header;
|
|
39
|
+
for (const h of hunks) {
|
|
40
|
+
if (cur !== header && tokensOf(cur) + tokensOf(h) > budgetTokens) {
|
|
41
|
+
parts.push(cur);
|
|
42
|
+
cur = header;
|
|
43
|
+
}
|
|
44
|
+
cur += h;
|
|
45
|
+
}
|
|
46
|
+
if (cur !== header)
|
|
47
|
+
parts.push(cur);
|
|
48
|
+
return parts;
|
|
49
|
+
}
|
|
50
|
+
export function planChunks(files, contexts = {}, targetTokens = CHUNK_TOKEN_TARGET) {
|
|
51
|
+
const units = [];
|
|
52
|
+
for (const f of groupByDir(files)) {
|
|
53
|
+
const patch = f.patch ?? '';
|
|
54
|
+
const ctxBlock = contexts[f.filename];
|
|
55
|
+
const ctxTokens = ctxBlock === undefined ? 0 : tokensOf(ctxBlock);
|
|
56
|
+
const section = (label, body) => {
|
|
57
|
+
const head = ctxBlock === undefined ? `### ${label}` : `### ${label}\n${ctxBlock}`;
|
|
58
|
+
return `${head}\n\`\`\`diff\n${body}\n\`\`\``;
|
|
59
|
+
};
|
|
60
|
+
const whole = tokensOf(patch) + ctxTokens + CHUNK_FILE_OVERHEAD;
|
|
61
|
+
if (whole <= targetTokens) {
|
|
62
|
+
units.push({ filename: f.filename, text: section(f.filename, patch), tokens: whole });
|
|
63
|
+
continue;
|
|
64
|
+
}
|
|
65
|
+
const parts = splitPatch(patch, Math.max(1, targetTokens - ctxTokens - CHUNK_FILE_OVERHEAD));
|
|
66
|
+
parts.forEach((p, i) => {
|
|
67
|
+
const label = parts.length > 1 ? `${f.filename} (part ${i + 1}/${parts.length})` : f.filename;
|
|
68
|
+
units.push({
|
|
69
|
+
filename: f.filename,
|
|
70
|
+
text: section(label, p),
|
|
71
|
+
tokens: tokensOf(p) + ctxTokens + CHUNK_FILE_OVERHEAD,
|
|
72
|
+
});
|
|
73
|
+
});
|
|
74
|
+
}
|
|
75
|
+
const chunks = [];
|
|
76
|
+
let cur = [];
|
|
77
|
+
let curTokens = 0;
|
|
78
|
+
const flush = () => {
|
|
79
|
+
if (cur.length === 0)
|
|
80
|
+
return;
|
|
81
|
+
chunks.push({
|
|
82
|
+
text: cur.map((u) => u.text).join('\n\n'),
|
|
83
|
+
files: [...new Set(cur.map((u) => u.filename))],
|
|
84
|
+
});
|
|
85
|
+
cur = [];
|
|
86
|
+
curTokens = 0;
|
|
87
|
+
};
|
|
88
|
+
for (const u of units) {
|
|
89
|
+
if (cur.length > 0 && curTokens + u.tokens > targetTokens)
|
|
90
|
+
flush();
|
|
91
|
+
cur.push(u);
|
|
92
|
+
curTokens += u.tokens;
|
|
93
|
+
}
|
|
94
|
+
flush();
|
|
95
|
+
return chunks;
|
|
96
|
+
}
|
|
@@ -18,6 +18,24 @@ export interface JsonSchema {
|
|
|
18
18
|
schema: Record<string, unknown>;
|
|
19
19
|
strict?: boolean;
|
|
20
20
|
}
|
|
21
|
+
/** One request inside an async batch; `customId` maps the result back. */
|
|
22
|
+
export interface BatchRequest {
|
|
23
|
+
customId: string;
|
|
24
|
+
messages: Message[];
|
|
25
|
+
schema?: JsonSchema;
|
|
26
|
+
provider?: ProviderRules;
|
|
27
|
+
}
|
|
28
|
+
/** Exactly one of `result` / `error` is set. */
|
|
29
|
+
export interface BatchItemResult {
|
|
30
|
+
customId: string;
|
|
31
|
+
result?: {
|
|
32
|
+
id: string;
|
|
33
|
+
content: string;
|
|
34
|
+
cost: CallCost;
|
|
35
|
+
model: string;
|
|
36
|
+
};
|
|
37
|
+
error?: string;
|
|
38
|
+
}
|
|
21
39
|
export interface OpenRouterClientOptions {
|
|
22
40
|
apiKey: string;
|
|
23
41
|
fetch?: typeof fetch;
|
|
@@ -48,7 +66,7 @@ export declare class OpenRouterClient {
|
|
|
48
66
|
private _fetch;
|
|
49
67
|
private _timeoutMs;
|
|
50
68
|
private _trace;
|
|
51
|
-
private
|
|
69
|
+
private _extraHeaders;
|
|
52
70
|
private _onCall;
|
|
53
71
|
constructor(opts: OpenRouterClientOptions);
|
|
54
72
|
private _request;
|
|
@@ -65,9 +83,27 @@ export declare class OpenRouterClient {
|
|
|
65
83
|
cost: CallCost;
|
|
66
84
|
model: string;
|
|
67
85
|
}>;
|
|
86
|
+
/**
|
|
87
|
+
* Async Batch API: submit every request in one POST, poll until a terminal
|
|
88
|
+
* status, and map the inline results back by `custom_id`. Throws when the
|
|
89
|
+
* batch fails/expires/cancels or the poll deadline passes — callers fall
|
|
90
|
+
* back to realtime. A per-request error is returned, not thrown.
|
|
91
|
+
* `endpoint` and `model` are serialized before `requests`.
|
|
92
|
+
*/
|
|
93
|
+
completeBatch(opts: {
|
|
94
|
+
/** Base slug; a trailing `:batch` variant suffix is stripped. */
|
|
95
|
+
model: string;
|
|
96
|
+
requests: BatchRequest[];
|
|
97
|
+
kind?: CallKind;
|
|
98
|
+
pollIntervalMs?: number;
|
|
99
|
+
/** Total time to wait for a terminal status before throwing. */
|
|
100
|
+
deadlineMs: number;
|
|
101
|
+
}): Promise<BatchItemResult[]>;
|
|
68
102
|
reconcile(id: string): Promise<{
|
|
69
103
|
costUsd: number;
|
|
70
104
|
}>;
|
|
105
|
+
private _buildBody;
|
|
106
|
+
private _requestHeaders;
|
|
71
107
|
private _tryComplete;
|
|
72
108
|
private _toApiMessages;
|
|
73
109
|
private _extractContent;
|
|
@@ -7,12 +7,15 @@ import { makeCallCost } from './cost.js';
|
|
|
7
7
|
* review job for 90+ minutes on one socket.
|
|
8
8
|
*/
|
|
9
9
|
const REQUEST_TIMEOUT_MS = 120_000;
|
|
10
|
+
const BATCH_TERMINAL = new Set(['completed', 'failed', 'expired', 'cancelled']);
|
|
11
|
+
/** Default poll cadence; a real batch probe took about six minutes. */
|
|
12
|
+
const BATCH_POLL_INTERVAL_MS = 10_000;
|
|
10
13
|
export class OpenRouterClient {
|
|
11
14
|
_apiKey;
|
|
12
15
|
_fetch;
|
|
13
16
|
_timeoutMs;
|
|
14
17
|
_trace;
|
|
15
|
-
|
|
18
|
+
_extraHeaders;
|
|
16
19
|
_onCall;
|
|
17
20
|
constructor(opts) {
|
|
18
21
|
if (!opts.apiKey) {
|
|
@@ -22,7 +25,7 @@ export class OpenRouterClient {
|
|
|
22
25
|
this._fetch = opts.fetch ?? globalThis.fetch;
|
|
23
26
|
this._timeoutMs = opts.timeoutMs ?? REQUEST_TIMEOUT_MS;
|
|
24
27
|
this._trace = opts.trace;
|
|
25
|
-
this.
|
|
28
|
+
this._extraHeaders = opts.headers;
|
|
26
29
|
this._onCall = opts.onCall;
|
|
27
30
|
}
|
|
28
31
|
_request(url, init) {
|
|
@@ -80,6 +83,97 @@ export class OpenRouterClient {
|
|
|
80
83
|
}
|
|
81
84
|
throw new Error(`${prefix}: ${errors.map((e) => e.message).join('; ')}`);
|
|
82
85
|
}
|
|
86
|
+
/**
|
|
87
|
+
* Async Batch API: submit every request in one POST, poll until a terminal
|
|
88
|
+
* status, and map the inline results back by `custom_id`. Throws when the
|
|
89
|
+
* batch fails/expires/cancels or the poll deadline passes — callers fall
|
|
90
|
+
* back to realtime. A per-request error is returned, not thrown.
|
|
91
|
+
* `endpoint` and `model` are serialized before `requests`.
|
|
92
|
+
*/
|
|
93
|
+
async completeBatch(opts) {
|
|
94
|
+
const kind = opts.kind ?? 'code';
|
|
95
|
+
const model = opts.model.replace(/:batch$/, '');
|
|
96
|
+
const submit = {
|
|
97
|
+
endpoint: '/v1/chat/completions',
|
|
98
|
+
model,
|
|
99
|
+
requests: opts.requests.map((r) => ({
|
|
100
|
+
custom_id: r.customId,
|
|
101
|
+
body: this._buildBody({
|
|
102
|
+
model,
|
|
103
|
+
messages: r.messages,
|
|
104
|
+
...(r.schema !== undefined ? { schema: r.schema } : {}),
|
|
105
|
+
...(r.provider !== undefined ? { provider: r.provider } : {}),
|
|
106
|
+
}),
|
|
107
|
+
})),
|
|
108
|
+
};
|
|
109
|
+
const started = Date.now();
|
|
110
|
+
const post = await this._request('https://openrouter.ai/api/v1/batches', {
|
|
111
|
+
method: 'POST',
|
|
112
|
+
headers: this._requestHeaders(),
|
|
113
|
+
body: JSON.stringify(submit),
|
|
114
|
+
});
|
|
115
|
+
if (!post.ok) {
|
|
116
|
+
throw (classifyHttpStatus(post.status, post.headers) ??
|
|
117
|
+
new Error(`OpenRouter batch submit failed: ${post.status} ${post.statusText}`));
|
|
118
|
+
}
|
|
119
|
+
let batch = unwrapBatch(await post.json());
|
|
120
|
+
const id = batch.id;
|
|
121
|
+
if (typeof id !== 'string' || id === '')
|
|
122
|
+
throw new Error('OpenRouter batch submit returned no id');
|
|
123
|
+
const interval = opts.pollIntervalMs ?? BATCH_POLL_INTERVAL_MS;
|
|
124
|
+
while (!BATCH_TERMINAL.has(String(batch.status))) {
|
|
125
|
+
if (Date.now() - started + interval > opts.deadlineMs) {
|
|
126
|
+
throw new Error(`OpenRouter batch ${id} not finished at the ${Math.round(opts.deadlineMs / 1000)}s poll deadline (status ${String(batch.status)})`);
|
|
127
|
+
}
|
|
128
|
+
await new Promise((r) => setTimeout(r, interval));
|
|
129
|
+
const res = await this._request(`https://openrouter.ai/api/v1/batches/${encodeURIComponent(id)}`, {
|
|
130
|
+
method: 'GET',
|
|
131
|
+
headers: this._requestHeaders(),
|
|
132
|
+
});
|
|
133
|
+
if (!res.ok) {
|
|
134
|
+
throw (classifyHttpStatus(res.status, res.headers) ??
|
|
135
|
+
new Error(`OpenRouter batch poll failed: ${res.status} ${res.statusText}`));
|
|
136
|
+
}
|
|
137
|
+
batch = unwrapBatch(await res.json());
|
|
138
|
+
}
|
|
139
|
+
if (batch.status !== 'completed') {
|
|
140
|
+
throw new Error(`OpenRouter batch ${id} ended ${String(batch.status)}`);
|
|
141
|
+
}
|
|
142
|
+
const byId = new Map();
|
|
143
|
+
for (const row of Array.isArray(batch.results) ? batch.results : []) {
|
|
144
|
+
const r = row;
|
|
145
|
+
if (typeof r.custom_id === 'string')
|
|
146
|
+
byId.set(r.custom_id, r);
|
|
147
|
+
}
|
|
148
|
+
return opts.requests.map((req) => {
|
|
149
|
+
const row = byId.get(req.customId);
|
|
150
|
+
if (row === undefined)
|
|
151
|
+
return { customId: req.customId, error: 'no result returned for request' };
|
|
152
|
+
if (row.error !== undefined && row.error !== null) {
|
|
153
|
+
const e = row.error;
|
|
154
|
+
return { customId: req.customId, error: typeof e.message === 'string' ? e.message : JSON.stringify(row.error) };
|
|
155
|
+
}
|
|
156
|
+
// Tolerate both a bare completion and a {status_code, body} envelope.
|
|
157
|
+
const raw = row.response;
|
|
158
|
+
const response = (raw?.body !== undefined && typeof raw.body === 'object' ? raw.body : raw);
|
|
159
|
+
if (response === undefined || response.usage === undefined || !Array.isArray(response.choices)) {
|
|
160
|
+
return { customId: req.customId, error: 'malformed batch response' };
|
|
161
|
+
}
|
|
162
|
+
const cost = makeCallCost(response, kind);
|
|
163
|
+
this._onCall?.({
|
|
164
|
+
id: response.id,
|
|
165
|
+
model: response.model,
|
|
166
|
+
kind,
|
|
167
|
+
costUsd: cost.costUsd,
|
|
168
|
+
tokens: cost.tokens,
|
|
169
|
+
...(this._trace ? { trace: this._trace } : {}),
|
|
170
|
+
});
|
|
171
|
+
return {
|
|
172
|
+
customId: req.customId,
|
|
173
|
+
result: { id: response.id, content: this._extractContent(response), cost, model: response.model },
|
|
174
|
+
};
|
|
175
|
+
});
|
|
176
|
+
}
|
|
83
177
|
async reconcile(id) {
|
|
84
178
|
const res = await this._request(`https://openrouter.ai/api/v1/generation?id=${encodeURIComponent(id)}`, {
|
|
85
179
|
method: 'GET',
|
|
@@ -92,7 +186,7 @@ export class OpenRouterClient {
|
|
|
92
186
|
const total = data.data?.total_cost ?? data.total_cost ?? 0;
|
|
93
187
|
return { costUsd: total };
|
|
94
188
|
}
|
|
95
|
-
|
|
189
|
+
_buildBody(req) {
|
|
96
190
|
const body = {
|
|
97
191
|
model: req.model,
|
|
98
192
|
messages: this._toApiMessages(req.messages),
|
|
@@ -116,13 +210,20 @@ export class OpenRouterClient {
|
|
|
116
210
|
if (this._trace) {
|
|
117
211
|
body.trace = this._trace;
|
|
118
212
|
}
|
|
119
|
-
|
|
213
|
+
return body;
|
|
214
|
+
}
|
|
215
|
+
_requestHeaders() {
|
|
216
|
+
return {
|
|
120
217
|
Authorization: `Bearer ${this._apiKey}`,
|
|
121
218
|
'Content-Type': 'application/json',
|
|
122
219
|
'X-Title': 'argus-reviewer',
|
|
123
220
|
'X-OpenRouter-Metadata': 'enabled',
|
|
124
|
-
...(this.
|
|
221
|
+
...(this._extraHeaders ?? {}),
|
|
125
222
|
};
|
|
223
|
+
}
|
|
224
|
+
async _tryComplete(req) {
|
|
225
|
+
const body = this._buildBody(req);
|
|
226
|
+
const headers = this._requestHeaders();
|
|
126
227
|
const res = await this._request('https://openrouter.ai/api/v1/chat/completions', {
|
|
127
228
|
method: 'POST',
|
|
128
229
|
headers,
|
|
@@ -158,3 +259,8 @@ export class OpenRouterClient {
|
|
|
158
259
|
return '';
|
|
159
260
|
}
|
|
160
261
|
}
|
|
262
|
+
function unwrapBatch(json) {
|
|
263
|
+
const o = (json ?? {});
|
|
264
|
+
const inner = o.data !== undefined && typeof o.data === 'object' && o.data !== null ? o.data : o;
|
|
265
|
+
return inner;
|
|
266
|
+
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "argus-reviewer-e2e",
|
|
3
|
-
"version": "0.4.
|
|
3
|
+
"version": "0.4.2",
|
|
4
4
|
"description": "argus-reviewer — open-source vision-model E2E testing harness. Record, replay, heal, and assert with any OpenRouter model under a hard budget cap.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"license": "MIT",
|
|
@@ -36,6 +36,7 @@
|
|
|
36
36
|
"build": "tsc -p tsconfig.build.json",
|
|
37
37
|
"check:dist": "test -z \"$(git status --porcelain -- dist/)\" && test -z \"$(git ls-files --others --ignored --exclude-standard -- dist/)\"",
|
|
38
38
|
"test": "vitest run",
|
|
39
|
+
"test:tmp-leaks": "node scripts/check-tmp-leaks.mjs",
|
|
39
40
|
"test:watch": "vitest",
|
|
40
41
|
"lint": "eslint .",
|
|
41
42
|
"format": "prettier --write .",
|