argus-reviewer-e2e 0.4.0 → 0.4.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,174 @@
1
+ /**
2
+ * Pure scaffold generator shared by `argus-reviewer init` and `init --pr`
3
+ * (and, later, the App's onboarding Worker, so the surfaces cannot drift).
4
+ * No I/O, no config execution, no environment reads. Workflow templates keep
5
+ * `persist-credentials: false`, a pinned action SHA, least-privilege
6
+ * `permissions`, and never `pull_request_target`.
7
+ */
8
+ /**
9
+ * The one place the user-facing action pin lives. Bump it here (and the golden
10
+ * fixtures) on release; see RELEASING.md. v0.4.0 and v0.4.1 ship an action.yml
11
+ * GitHub cannot parse, so never pin to them.
12
+ */
13
+ export const ACTION_PIN_SHA = 'dbf4b7f669507402d527cb0e350285ca56129448';
14
+ export const ACTION_PIN_TAG = 'v0.4.2';
15
+ const ACTION_PIN = `${ACTION_PIN_SHA} # ${ACTION_PIN_TAG}`;
16
+ export function initConfig(a0Host) {
17
+ // R19 — a detected Agent Zero host earns a labeled suggestion, never an
18
+ // enabled lane: `verify --a0` is explicit opt-in per run, and completed
19
+ // delegations cap at inconclusive (self-reported evidence).
20
+ const a0Block = a0Host !== undefined
21
+ ? `
22
+ // Optional: Agent Zero detected at ${a0Host}. Nothing below runs unless
23
+ // you ask for it — both stays commented until you opt in deliberately.
24
+ // a0: { url: ${JSON.stringify(a0Host)} }, // enables \`verify --a0\` (self-reported, unmetered)
25
+ // heal: 'a0', // escalates a failed heal to the A0 host
26
+ `
27
+ : '';
28
+ return `import { defineConfig } from 'argus-reviewer-e2e'
29
+
30
+ export default defineConfig({
31
+ // The app under test. command boots it (omit if it is already running);
32
+ // argus-reviewer polls url until it responds before running tests.
33
+ target: {
34
+ command: 'npm run dev',
35
+ url: 'http://localhost:3000',
36
+ readyTimeoutMs: 30_000,
37
+ },
38
+ // Hard per-run cap on vision-model spend (USD). Steps replayed from the
39
+ // fingerprint cache cost $0 regardless of this cap.
40
+ budgetUsd: 1,
41
+ testsDir: 'tests/argus',
42
+ // Exploratory lane: after the test loop, a bounded agent pass probes the
43
+ // app itself — same-origin navigation, clicks, invalid input — while taps
44
+ // capture console errors, page errors, and failed requests. Findings
45
+ // render as 'observed' — evidence only, never verdict-changing.
46
+ // maxSteps caps acts per run; budgetUsd caps explore model spend (falls
47
+ // back to budgetUsd). Point it at disposable targets only — clicks and
48
+ // form submits have real side effects.
49
+ // explore: { enabled: true, maxSteps: 20, budgetUsd: 0.25 },${a0Block}
50
+ })
51
+ `;
52
+ }
53
+ export const INIT_TEST = `test('home renders', async (td) => {
54
+ const ok = await td.assert('the page rendered without obvious errors')
55
+ if (!ok) throw new Error('home did not render')
56
+ })
57
+ `;
58
+ export const INIT_WORKFLOW = `name: argus-reviewer
59
+
60
+ on:
61
+ pull_request:
62
+ # 'labeled' lets a maintainer re-trigger with the argus-probe label when
63
+ # sandbox probes are enabled for fork PRs.
64
+ types: [opened, synchronize, reopened, labeled]
65
+
66
+ jobs:
67
+ argus:
68
+ runs-on: ubuntu-latest
69
+ # 'labeled' fires on EVERY label — only argus-probe is the fork-gate
70
+ # signal worth a full review run.
71
+ if: github.event.action != 'labeled' || github.event.label.name == 'argus-probe'
72
+ permissions:
73
+ contents: read
74
+ issues: write
75
+ pull-requests: write
76
+ checks: write
77
+ statuses: write
78
+ steps:
79
+ # persist-credentials: false keeps the GITHUB_TOKEN out of .git/config —
80
+ # the probe sandbox masks .git regardless, but don't store it at all.
81
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
82
+ with:
83
+ persist-credentials: false
84
+ ref: \${{ github.event.pull_request.head.sha || github.sha }}
85
+ # Optional verdict-as-review: let Argus submit APPROVE / REQUEST_CHANGES
86
+ # so require_approving_reviews counts it. GITHUB_TOKEN cannot approve, so
87
+ # create + install your own GitHub App (docs/github-app.md), set the
88
+ # ARGUS_APP_ID variable and ARGUS_APP_PRIVATE_KEY secret, then uncomment:
89
+ # - uses: actions/create-github-app-token@fee1f7d63c2ff003460e3d139729b119787bc349 # v2
90
+ # id: argus-app
91
+ # with:
92
+ # app-id: \${{ vars.ARGUS_APP_ID }}
93
+ # private-key: \${{ secrets.ARGUS_APP_PRIVATE_KEY }}
94
+ # and pass approval-token plus its evidence inputs to the action below:
95
+ # approval-token: \${{ steps.argus-app.outputs.token }}
96
+ # approval-evidence: 'npm test' # command the approval stands on
97
+ # approval-check: 'test' # check-run name, green on head SHA
98
+ - uses: duketopceo/Argus/action@@@ACTION_PIN@@
99
+ with:
100
+ openrouter-api-key: \${{ secrets.OPENROUTER_API_KEY }}
101
+ `.replace('@@ACTION_PIN@@', ACTION_PIN);
102
+ export const INIT_MENTION_WORKFLOW = `name: argus-mention
103
+
104
+ # @argus mention commands on PR comments — '@argus review', '@argus
105
+ # record "<flow>"', '@argus persist', '@argus help'. issue_comment is
106
+ # strictly more privileged than pull_request (secrets + write token are
107
+ # present), so the checkout below deliberately resolves the BASE ref —
108
+ # never the PR head. Argus reviews the head diff over the API.
109
+ on:
110
+ issue_comment:
111
+ types: [created]
112
+
113
+ jobs:
114
+ argus-mention:
115
+ runs-on: ubuntu-latest
116
+ if: github.event.issue.pull_request && startsWith(github.event.comment.body, '@argus')
117
+ permissions:
118
+ # contents: write — '@argus persist' commits reproduced probes to an
119
+ # argus/ branch via the git/refs + contents APIs and opens a PR.
120
+ contents: write
121
+ issues: write
122
+ pull-requests: write
123
+ checks: write
124
+ statuses: write
125
+ steps:
126
+ # No 'ref' — the default checkout resolves the base branch. persist
127
+ # writes via the API, so checkout credentials stay disabled.
128
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7
129
+ with:
130
+ persist-credentials: false
131
+ # Record commands need the app's dependencies to boot its target.
132
+ # Uncomment if you use '@argus record':
133
+ # - run: npm ci
134
+ - uses: duketopceo/Argus/action@@@ACTION_PIN@@
135
+ with:
136
+ openrouter-api-key: \${{ secrets.OPENROUTER_API_KEY }}
137
+ # '@argus record' uploads the generated test + flow cache as an
138
+ # artifact — committing to a PR branch is intentionally not done.
139
+ - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
140
+ if: contains(github.event.comment.body, 'record')
141
+ with:
142
+ name: argus-recorded-flow
143
+ path: |
144
+ tests/argus/
145
+ .argus-reviewer-cache/
146
+ if-no-files-found: ignore
147
+ `.replace('@@ACTION_PIN@@', ACTION_PIN);
148
+ export const CONFIG_PATH = 'argus-reviewer.config.ts';
149
+ /** The files `init` writes, in write order. */
150
+ export function renderScaffold(opts) {
151
+ const files = [
152
+ { path: 'tests/argus/smoke.test.ts', content: INIT_TEST },
153
+ { path: '.github/workflows/argus-reviewer.yml', content: INIT_WORKFLOW },
154
+ { path: '.github/workflows/argus-mention.yml', content: INIT_MENTION_WORKFLOW },
155
+ ];
156
+ if (opts.includeConfig)
157
+ files.unshift({ path: CONFIG_PATH, content: initConfig(opts.a0Host) });
158
+ return files;
159
+ }
160
+ /**
161
+ * "What runs and what it costs": what is sent to the provider, the default
162
+ * budget, and the stop path. Kept verbatim (DESIGN 7.8). `budgetUsd` comes
163
+ * from the caller's resolved defaults so it cannot go stale here.
164
+ */
165
+ export function scaffoldChecklist(budgetUsd) {
166
+ return [
167
+ 'What runs and what it costs:',
168
+ ' sent to provider PR diffs, page screenshots/DOM snapshots, and',
169
+ ' review prompts — via your OpenRouter key (BYOK)',
170
+ ` default budget $${budgetUsd}/run cap (budgetUsd); cached replay costs $0`,
171
+ ' how to stop Ctrl+C locally; in CI remove the workflow file',
172
+ ' or delete the OPENROUTER_API_KEY secret',
173
+ ];
174
+ }
@@ -10,3 +10,16 @@ export declare function overLimit(spentUsd: number, limitUsd: number): boolean;
10
10
  export declare function addProviderCalls(budget: BudgetSummary, calls: CallCost[] | undefined): BudgetSummary;
11
11
  export declare function addA0Task(budget: BudgetSummary, elapsedMs: number, metered: boolean): BudgetSummary;
12
12
  export declare function budgetCanSpend(budget: BudgetSummary, nextCostUsd: number): boolean;
13
+ export declare function estimateRequestCostUsd(req: {
14
+ messages: unknown;
15
+ schema?: unknown;
16
+ }): number;
17
+ /**
18
+ * How many leading requests of a batch fit in the remaining budget. A batch
19
+ * cannot be cancelled once submitted, so the guard sizes it up front.
20
+ * `limitUsd` undefined = unlimited.
21
+ */
22
+ export declare function affordableBatchPrefix(requests: ReadonlyArray<{
23
+ messages: unknown;
24
+ schema?: unknown;
25
+ }>, limitUsd: number | undefined, spentUsd: number): number;
@@ -46,3 +46,37 @@ export function budgetCanSpend(budget, nextCostUsd) {
46
46
  (budget.limitUsd === undefined ||
47
47
  budget.spentUsd + nextCostUsd <= budget.limitUsd + USD_EPSILON));
48
48
  }
49
+ /**
50
+ * Conservative per-request cost ceiling (USD) used ONLY where a paid call
51
+ * cannot be stopped mid-flight (batch submission). Argus has no live price
52
+ * table, so this prices tokens at a deliberately high blended rate (well above
53
+ * the default review models) and reserves a fixed output allowance. Real
54
+ * spend is metered from provider usage; this only decides whether to submit.
55
+ */
56
+ const EST_INPUT_USD_PER_TOKEN = 3 / 1_000_000;
57
+ const EST_OUTPUT_USD_PER_TOKEN = 12 / 1_000_000;
58
+ const EST_OUTPUT_TOKENS = 4_000;
59
+ const EST_CHARS_PER_TOKEN = 3;
60
+ export function estimateRequestCostUsd(req) {
61
+ const chars = JSON.stringify(req.messages ?? '').length + JSON.stringify(req.schema ?? '').length;
62
+ const inputTokens = Math.ceil(chars / EST_CHARS_PER_TOKEN);
63
+ return inputTokens * EST_INPUT_USD_PER_TOKEN + EST_OUTPUT_TOKENS * EST_OUTPUT_USD_PER_TOKEN;
64
+ }
65
+ /**
66
+ * How many leading requests of a batch fit in the remaining budget. A batch
67
+ * cannot be cancelled once submitted, so the guard sizes it up front.
68
+ * `limitUsd` undefined = unlimited.
69
+ */
70
+ export function affordableBatchPrefix(requests, limitUsd, spentUsd) {
71
+ if (limitUsd === undefined)
72
+ return requests.length;
73
+ let projected = spentUsd;
74
+ let n = 0;
75
+ for (const r of requests) {
76
+ projected += estimateRequestCostUsd(r);
77
+ if (projected > limitUsd + USD_EPSILON)
78
+ break;
79
+ n++;
80
+ }
81
+ return n;
82
+ }
@@ -0,0 +1,21 @@
1
+ /**
2
+ * Chunk planning for large PRs.
3
+ *
4
+ * A diff over the per-call token target is reviewed in several model calls
5
+ * instead of one oversized (or truncated) prompt. Files are grouped by
6
+ * directory so a chunk reads as one area of the change, and a single patch
7
+ * larger than the target is split at hunk boundaries. Every chunk records
8
+ * which files it carries so a partial review can say what it did not cover.
9
+ */
10
+ export declare const CHUNK_TOKEN_TARGET = 6000;
11
+ export interface ChunkFile {
12
+ filename: string;
13
+ patch?: string;
14
+ }
15
+ export interface PlannedChunk {
16
+ /** Prompt text for this chunk. */
17
+ text: string;
18
+ /** Files with content in this chunk (a split file appears in each part). */
19
+ files: string[];
20
+ }
21
+ export declare function planChunks(files: ChunkFile[], contexts?: Record<string, string>, targetTokens?: number): PlannedChunk[];
@@ -0,0 +1,96 @@
1
+ /**
2
+ * Chunk planning for large PRs.
3
+ *
4
+ * A diff over the per-call token target is reviewed in several model calls
5
+ * instead of one oversized (or truncated) prompt. Files are grouped by
6
+ * directory so a chunk reads as one area of the change, and a single patch
7
+ * larger than the target is split at hunk boundaries. Every chunk records
8
+ * which files it carries so a partial review can say what it did not cover.
9
+ */
10
+ export const CHUNK_TOKEN_TARGET = 6000;
11
+ const CHUNK_FILE_OVERHEAD = 100;
12
+ const tokensOf = (s) => Math.ceil(s.length / 4);
13
+ function dirOf(filename) {
14
+ const i = filename.lastIndexOf('/');
15
+ return i < 0 ? '' : filename.slice(0, i);
16
+ }
17
+ /** Stable group-by-directory; groups keep first-appearance order. */
18
+ function groupByDir(files) {
19
+ const groups = new Map();
20
+ for (const f of files) {
21
+ const d = dirOf(f.filename);
22
+ const g = groups.get(d);
23
+ if (g === undefined)
24
+ groups.set(d, [f]);
25
+ else
26
+ g.push(f);
27
+ }
28
+ return [...groups.values()].flat();
29
+ }
30
+ /** Split a patch at `@@` hunk starts into parts that each fit `budgetTokens`. */
31
+ function splitPatch(patch, budgetTokens) {
32
+ const firstHunk = patch.search(/^@@/m);
33
+ if (firstHunk < 0)
34
+ return [patch];
35
+ const header = patch.slice(0, firstHunk);
36
+ const hunks = patch.slice(firstHunk).split(/^(?=@@)/m);
37
+ const parts = [];
38
+ let cur = header;
39
+ for (const h of hunks) {
40
+ if (cur !== header && tokensOf(cur) + tokensOf(h) > budgetTokens) {
41
+ parts.push(cur);
42
+ cur = header;
43
+ }
44
+ cur += h;
45
+ }
46
+ if (cur !== header)
47
+ parts.push(cur);
48
+ return parts;
49
+ }
50
+ export function planChunks(files, contexts = {}, targetTokens = CHUNK_TOKEN_TARGET) {
51
+ const units = [];
52
+ for (const f of groupByDir(files)) {
53
+ const patch = f.patch ?? '';
54
+ const ctxBlock = contexts[f.filename];
55
+ const ctxTokens = ctxBlock === undefined ? 0 : tokensOf(ctxBlock);
56
+ const section = (label, body) => {
57
+ const head = ctxBlock === undefined ? `### ${label}` : `### ${label}\n${ctxBlock}`;
58
+ return `${head}\n\`\`\`diff\n${body}\n\`\`\``;
59
+ };
60
+ const whole = tokensOf(patch) + ctxTokens + CHUNK_FILE_OVERHEAD;
61
+ if (whole <= targetTokens) {
62
+ units.push({ filename: f.filename, text: section(f.filename, patch), tokens: whole });
63
+ continue;
64
+ }
65
+ const parts = splitPatch(patch, Math.max(1, targetTokens - ctxTokens - CHUNK_FILE_OVERHEAD));
66
+ parts.forEach((p, i) => {
67
+ const label = parts.length > 1 ? `${f.filename} (part ${i + 1}/${parts.length})` : f.filename;
68
+ units.push({
69
+ filename: f.filename,
70
+ text: section(label, p),
71
+ tokens: tokensOf(p) + ctxTokens + CHUNK_FILE_OVERHEAD,
72
+ });
73
+ });
74
+ }
75
+ const chunks = [];
76
+ let cur = [];
77
+ let curTokens = 0;
78
+ const flush = () => {
79
+ if (cur.length === 0)
80
+ return;
81
+ chunks.push({
82
+ text: cur.map((u) => u.text).join('\n\n'),
83
+ files: [...new Set(cur.map((u) => u.filename))],
84
+ });
85
+ cur = [];
86
+ curTokens = 0;
87
+ };
88
+ for (const u of units) {
89
+ if (cur.length > 0 && curTokens + u.tokens > targetTokens)
90
+ flush();
91
+ cur.push(u);
92
+ curTokens += u.tokens;
93
+ }
94
+ flush();
95
+ return chunks;
96
+ }
@@ -18,6 +18,24 @@ export interface JsonSchema {
18
18
  schema: Record<string, unknown>;
19
19
  strict?: boolean;
20
20
  }
21
+ /** One request inside an async batch; `customId` maps the result back. */
22
+ export interface BatchRequest {
23
+ customId: string;
24
+ messages: Message[];
25
+ schema?: JsonSchema;
26
+ provider?: ProviderRules;
27
+ }
28
+ /** Exactly one of `result` / `error` is set. */
29
+ export interface BatchItemResult {
30
+ customId: string;
31
+ result?: {
32
+ id: string;
33
+ content: string;
34
+ cost: CallCost;
35
+ model: string;
36
+ };
37
+ error?: string;
38
+ }
21
39
  export interface OpenRouterClientOptions {
22
40
  apiKey: string;
23
41
  fetch?: typeof fetch;
@@ -48,7 +66,7 @@ export declare class OpenRouterClient {
48
66
  private _fetch;
49
67
  private _timeoutMs;
50
68
  private _trace;
51
- private _headers;
69
+ private _extraHeaders;
52
70
  private _onCall;
53
71
  constructor(opts: OpenRouterClientOptions);
54
72
  private _request;
@@ -65,9 +83,27 @@ export declare class OpenRouterClient {
65
83
  cost: CallCost;
66
84
  model: string;
67
85
  }>;
86
+ /**
87
+ * Async Batch API: submit every request in one POST, poll until a terminal
88
+ * status, and map the inline results back by `custom_id`. Throws when the
89
+ * batch fails/expires/cancels or the poll deadline passes — callers fall
90
+ * back to realtime. A per-request error is returned, not thrown.
91
+ * `endpoint` and `model` are serialized before `requests`.
92
+ */
93
+ completeBatch(opts: {
94
+ /** Base slug; a trailing `:batch` variant suffix is stripped. */
95
+ model: string;
96
+ requests: BatchRequest[];
97
+ kind?: CallKind;
98
+ pollIntervalMs?: number;
99
+ /** Total time to wait for a terminal status before throwing. */
100
+ deadlineMs: number;
101
+ }): Promise<BatchItemResult[]>;
68
102
  reconcile(id: string): Promise<{
69
103
  costUsd: number;
70
104
  }>;
105
+ private _buildBody;
106
+ private _requestHeaders;
71
107
  private _tryComplete;
72
108
  private _toApiMessages;
73
109
  private _extractContent;
@@ -7,12 +7,15 @@ import { makeCallCost } from './cost.js';
7
7
  * review job for 90+ minutes on one socket.
8
8
  */
9
9
  const REQUEST_TIMEOUT_MS = 120_000;
10
+ const BATCH_TERMINAL = new Set(['completed', 'failed', 'expired', 'cancelled']);
11
+ /** Default poll cadence; a real batch probe took about six minutes. */
12
+ const BATCH_POLL_INTERVAL_MS = 10_000;
10
13
  export class OpenRouterClient {
11
14
  _apiKey;
12
15
  _fetch;
13
16
  _timeoutMs;
14
17
  _trace;
15
- _headers;
18
+ _extraHeaders;
16
19
  _onCall;
17
20
  constructor(opts) {
18
21
  if (!opts.apiKey) {
@@ -22,7 +25,7 @@ export class OpenRouterClient {
22
25
  this._fetch = opts.fetch ?? globalThis.fetch;
23
26
  this._timeoutMs = opts.timeoutMs ?? REQUEST_TIMEOUT_MS;
24
27
  this._trace = opts.trace;
25
- this._headers = opts.headers;
28
+ this._extraHeaders = opts.headers;
26
29
  this._onCall = opts.onCall;
27
30
  }
28
31
  _request(url, init) {
@@ -80,6 +83,97 @@ export class OpenRouterClient {
80
83
  }
81
84
  throw new Error(`${prefix}: ${errors.map((e) => e.message).join('; ')}`);
82
85
  }
86
+ /**
87
+ * Async Batch API: submit every request in one POST, poll until a terminal
88
+ * status, and map the inline results back by `custom_id`. Throws when the
89
+ * batch fails/expires/cancels or the poll deadline passes — callers fall
90
+ * back to realtime. A per-request error is returned, not thrown.
91
+ * `endpoint` and `model` are serialized before `requests`.
92
+ */
93
+ async completeBatch(opts) {
94
+ const kind = opts.kind ?? 'code';
95
+ const model = opts.model.replace(/:batch$/, '');
96
+ const submit = {
97
+ endpoint: '/v1/chat/completions',
98
+ model,
99
+ requests: opts.requests.map((r) => ({
100
+ custom_id: r.customId,
101
+ body: this._buildBody({
102
+ model,
103
+ messages: r.messages,
104
+ ...(r.schema !== undefined ? { schema: r.schema } : {}),
105
+ ...(r.provider !== undefined ? { provider: r.provider } : {}),
106
+ }),
107
+ })),
108
+ };
109
+ const started = Date.now();
110
+ const post = await this._request('https://openrouter.ai/api/v1/batches', {
111
+ method: 'POST',
112
+ headers: this._requestHeaders(),
113
+ body: JSON.stringify(submit),
114
+ });
115
+ if (!post.ok) {
116
+ throw (classifyHttpStatus(post.status, post.headers) ??
117
+ new Error(`OpenRouter batch submit failed: ${post.status} ${post.statusText}`));
118
+ }
119
+ let batch = unwrapBatch(await post.json());
120
+ const id = batch.id;
121
+ if (typeof id !== 'string' || id === '')
122
+ throw new Error('OpenRouter batch submit returned no id');
123
+ const interval = opts.pollIntervalMs ?? BATCH_POLL_INTERVAL_MS;
124
+ while (!BATCH_TERMINAL.has(String(batch.status))) {
125
+ if (Date.now() - started + interval > opts.deadlineMs) {
126
+ throw new Error(`OpenRouter batch ${id} not finished at the ${Math.round(opts.deadlineMs / 1000)}s poll deadline (status ${String(batch.status)})`);
127
+ }
128
+ await new Promise((r) => setTimeout(r, interval));
129
+ const res = await this._request(`https://openrouter.ai/api/v1/batches/${encodeURIComponent(id)}`, {
130
+ method: 'GET',
131
+ headers: this._requestHeaders(),
132
+ });
133
+ if (!res.ok) {
134
+ throw (classifyHttpStatus(res.status, res.headers) ??
135
+ new Error(`OpenRouter batch poll failed: ${res.status} ${res.statusText}`));
136
+ }
137
+ batch = unwrapBatch(await res.json());
138
+ }
139
+ if (batch.status !== 'completed') {
140
+ throw new Error(`OpenRouter batch ${id} ended ${String(batch.status)}`);
141
+ }
142
+ const byId = new Map();
143
+ for (const row of Array.isArray(batch.results) ? batch.results : []) {
144
+ const r = row;
145
+ if (typeof r.custom_id === 'string')
146
+ byId.set(r.custom_id, r);
147
+ }
148
+ return opts.requests.map((req) => {
149
+ const row = byId.get(req.customId);
150
+ if (row === undefined)
151
+ return { customId: req.customId, error: 'no result returned for request' };
152
+ if (row.error !== undefined && row.error !== null) {
153
+ const e = row.error;
154
+ return { customId: req.customId, error: typeof e.message === 'string' ? e.message : JSON.stringify(row.error) };
155
+ }
156
+ // Tolerate both a bare completion and a {status_code, body} envelope.
157
+ const raw = row.response;
158
+ const response = (raw?.body !== undefined && typeof raw.body === 'object' ? raw.body : raw);
159
+ if (response === undefined || response.usage === undefined || !Array.isArray(response.choices)) {
160
+ return { customId: req.customId, error: 'malformed batch response' };
161
+ }
162
+ const cost = makeCallCost(response, kind);
163
+ this._onCall?.({
164
+ id: response.id,
165
+ model: response.model,
166
+ kind,
167
+ costUsd: cost.costUsd,
168
+ tokens: cost.tokens,
169
+ ...(this._trace ? { trace: this._trace } : {}),
170
+ });
171
+ return {
172
+ customId: req.customId,
173
+ result: { id: response.id, content: this._extractContent(response), cost, model: response.model },
174
+ };
175
+ });
176
+ }
83
177
  async reconcile(id) {
84
178
  const res = await this._request(`https://openrouter.ai/api/v1/generation?id=${encodeURIComponent(id)}`, {
85
179
  method: 'GET',
@@ -92,7 +186,7 @@ export class OpenRouterClient {
92
186
  const total = data.data?.total_cost ?? data.total_cost ?? 0;
93
187
  return { costUsd: total };
94
188
  }
95
- async _tryComplete(req) {
189
+ _buildBody(req) {
96
190
  const body = {
97
191
  model: req.model,
98
192
  messages: this._toApiMessages(req.messages),
@@ -116,13 +210,20 @@ export class OpenRouterClient {
116
210
  if (this._trace) {
117
211
  body.trace = this._trace;
118
212
  }
119
- const headers = {
213
+ return body;
214
+ }
215
+ _requestHeaders() {
216
+ return {
120
217
  Authorization: `Bearer ${this._apiKey}`,
121
218
  'Content-Type': 'application/json',
122
219
  'X-Title': 'argus-reviewer',
123
220
  'X-OpenRouter-Metadata': 'enabled',
124
- ...(this._headers ?? {}),
221
+ ...(this._extraHeaders ?? {}),
125
222
  };
223
+ }
224
+ async _tryComplete(req) {
225
+ const body = this._buildBody(req);
226
+ const headers = this._requestHeaders();
126
227
  const res = await this._request('https://openrouter.ai/api/v1/chat/completions', {
127
228
  method: 'POST',
128
229
  headers,
@@ -158,3 +259,8 @@ export class OpenRouterClient {
158
259
  return '';
159
260
  }
160
261
  }
262
+ function unwrapBatch(json) {
263
+ const o = (json ?? {});
264
+ const inner = o.data !== undefined && typeof o.data === 'object' && o.data !== null ? o.data : o;
265
+ return inner;
266
+ }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "argus-reviewer-e2e",
3
- "version": "0.4.0",
3
+ "version": "0.4.2",
4
4
  "description": "argus-reviewer — open-source vision-model E2E testing harness. Record, replay, heal, and assert with any OpenRouter model under a hard budget cap.",
5
5
  "type": "module",
6
6
  "license": "MIT",
@@ -36,6 +36,7 @@
36
36
  "build": "tsc -p tsconfig.build.json",
37
37
  "check:dist": "test -z \"$(git status --porcelain -- dist/)\" && test -z \"$(git ls-files --others --ignored --exclude-standard -- dist/)\"",
38
38
  "test": "vitest run",
39
+ "test:tmp-leaks": "node scripts/check-tmp-leaks.mjs",
39
40
  "test:watch": "vitest",
40
41
  "lint": "eslint .",
41
42
  "format": "prettier --write .",