@amenophis1er/foreman 0.1.16 → 0.1.18
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +74 -3
- package/package.json +1 -1
- package/src/crew.test.ts +376 -0
- package/src/crew.ts +330 -0
- package/src/fleet-planner.test.ts +39 -1
- package/src/fleet-planner.ts +83 -3
- package/src/gitwork.test.ts +133 -2
- package/src/gitwork.ts +211 -2
- package/src/mcp.test.ts +38 -1
- package/src/mcp.ts +86 -5
- package/src/notify/commands.test.ts +2 -0
- package/src/notify/commands.ts +6 -0
- package/src/notify/telegram.ts +1 -0
- package/src/notify.test.ts +61 -0
- package/src/notify.ts +45 -1
- package/src/orchestrator.test.ts +516 -2
- package/src/orchestrator.ts +514 -69
- package/src/run-crew.test.ts +99 -0
- package/src/run-crew.ts +101 -0
- package/src/schedule-guards.test.ts +236 -0
- package/src/schedule-guards.ts +149 -0
- package/src/schedule.test.ts +240 -0
- package/src/schedule.ts +343 -0
- package/src/server.ts +674 -11
- package/src/store.test.ts +81 -1
- package/src/store.ts +117 -3
- package/src/types.ts +92 -0
- package/ui/dist/assets/index-0QuGXbFg.js +76 -0
- package/ui/dist/index.html +1 -1
- package/ui/dist/assets/index-DOVnExqF.js +0 -68
package/src/notify.test.ts
CHANGED
|
@@ -123,6 +123,67 @@ test('a transport that throws costs a failure count, never an exception on the e
|
|
|
123
123
|
assert.equal(hub.delivered, 0);
|
|
124
124
|
});
|
|
125
125
|
|
|
126
|
+
test('a paused schedule says which one and why, and is gated by its cause', () => {
|
|
127
|
+
const c = ctx()();
|
|
128
|
+
const cap = shape(env('schedule_paused', { scheduleId: 's1', name: 'nightly deps', reason: 'monthly-cap' }), c)!;
|
|
129
|
+
assert.equal(cap.key, 'schedule:s1');
|
|
130
|
+
assert.equal(cap.gate, 'budget', 'the monthly ceiling is a spending guard');
|
|
131
|
+
assert.match(cap.text, /nightly deps/);
|
|
132
|
+
assert.match(cap.text, /this month's scheduled spend would go past the ceiling|this month's scheduled spend would go past the ceiling/);
|
|
133
|
+
|
|
134
|
+
const fail = shape(env('schedule_paused', { scheduleId: 's1', name: 'nightly deps', reason: 'failures' }), c)!;
|
|
135
|
+
assert.equal(fail.key, 'schedule:s1');
|
|
136
|
+
assert.equal(fail.gate, 'needsYou', 'only a human can resume it');
|
|
137
|
+
assert.match(fail.text, /two scheduled runs in a row failed/);
|
|
138
|
+
|
|
139
|
+
// Resuming is deliberately a dashboard act, so there is nothing to tap.
|
|
140
|
+
for (const s of [cap, fail]) {
|
|
141
|
+
assert.match(s.text, /Resume it from the Foreman dashboard/);
|
|
142
|
+
assert.equal(s.buttons, undefined);
|
|
143
|
+
}
|
|
144
|
+
});
|
|
145
|
+
|
|
146
|
+
test('an unknown pause reason still shapes, and does not throw', () => {
|
|
147
|
+
const c = ctx()();
|
|
148
|
+
assert.doesNotThrow(() => shape(env('schedule_paused', { scheduleId: 's1', name: 'nightly deps', reason: 'kaput' }), c));
|
|
149
|
+
const s = shape(env('schedule_paused', { scheduleId: 's1', name: 'nightly deps', reason: 'kaput' }), c)!;
|
|
150
|
+
assert.equal(s.key, 'schedule:s1');
|
|
151
|
+
assert.equal(s.gate, 'done');
|
|
152
|
+
assert.match(s.text, /nightly deps/);
|
|
153
|
+
assert.doesNotThrow(() => shape(env('schedule_skipped', { scheduleId: 's1', name: 'nightly deps', reason: 'moon phase' }), c));
|
|
154
|
+
});
|
|
155
|
+
|
|
156
|
+
test('a skipped scheduled run is informational, and consecutive skips are not deduped', () => {
|
|
157
|
+
const c = ctx()();
|
|
158
|
+
const s = shape(env('schedule_skipped', { scheduleId: 's1', name: 'nightly deps', reason: 'project busy' }), c)!;
|
|
159
|
+
assert.equal(s.gate, 'done');
|
|
160
|
+
assert.equal(s.key, 'skip:s1:1000000');
|
|
161
|
+
assert.match(s.text, /nightly deps/);
|
|
162
|
+
assert.match(s.text, /already had a mission running/);
|
|
163
|
+
assert.match(s.text, /next scheduled run stands/);
|
|
164
|
+
// Like a stall, the timestamp is in the key so tonight's skip is not
|
|
165
|
+
// swallowed by last night's.
|
|
166
|
+
const later = shape(env('schedule_skipped', { scheduleId: 's1', name: 'nightly deps', reason: 'project busy' }, { ts: 1_000_500 }), c)!;
|
|
167
|
+
assert.notEqual(later.key, s.key);
|
|
168
|
+
});
|
|
169
|
+
|
|
170
|
+
test('a scheduled run says nobody pressed start; an ordinary one reads exactly as before', () => {
|
|
171
|
+
const c = ctx()();
|
|
172
|
+
const plain = shape(env('run_finished', { status: 'done' }), c)!;
|
|
173
|
+
assert.equal(plain.text,
|
|
174
|
+
'<b>Mission done</b> · shop\n<i>Build the checkout page</i>' +
|
|
175
|
+
'\n<a href="http://box.local:4177/#/p/p1/r/r1">Open in Foreman</a>');
|
|
176
|
+
|
|
177
|
+
const sched = shape(env('run_finished', { status: 'done', scheduled: true, scheduleName: 'nightly deps' }), c)!;
|
|
178
|
+
assert.equal(sched.key, plain.key, 'same key, so it still edits and dedupes as before');
|
|
179
|
+
assert.equal(sched.gate, plain.gate);
|
|
180
|
+
assert.match(sched.text, /Mission done · scheduled · nightly deps/);
|
|
181
|
+
|
|
182
|
+
// Without a name it still marks itself as unattended.
|
|
183
|
+
const anon = shape(env('run_finished', { status: 'error', scheduled: true }), c)!;
|
|
184
|
+
assert.match(anon.text, /Mission failed · scheduled/);
|
|
185
|
+
});
|
|
186
|
+
|
|
126
187
|
// ---------------------------------------------------------------------------
|
|
127
188
|
// Telegram, against a stub of the Bot API
|
|
128
189
|
// ---------------------------------------------------------------------------
|
package/src/notify.ts
CHANGED
|
@@ -260,12 +260,56 @@ export function shape(env: Envelope, ctx: NotifyContext): Shaped | null {
|
|
|
260
260
|
case 'run_finished': {
|
|
261
261
|
const status = String(d.status ?? 'done');
|
|
262
262
|
const title = status === 'done' ? 'Mission done' : status === 'error' ? 'Mission failed' : 'Mission interrupted';
|
|
263
|
+
// A scheduled run says so in its title: nobody pressed start, so the
|
|
264
|
+
// first question a phone raises — "who did this?" — is already answered.
|
|
265
|
+
const sched = d.scheduled
|
|
266
|
+
? ` · scheduled${d.scheduleName ? ` · ${clip(d.scheduleName, 40)}` : ''}`
|
|
267
|
+
: '';
|
|
263
268
|
return { key: `finished:${env.runId}`, gate: 'done',
|
|
264
|
-
text: `${head(title)}${runLine}${foot}` };
|
|
269
|
+
text: `${head(title + sched)}${runLine}${foot}` };
|
|
265
270
|
}
|
|
266
271
|
case 'run_error':
|
|
267
272
|
return { key: `error:${env.runId}:${env.ts ?? ''}`, gate: 'done',
|
|
268
273
|
text: `${head('Mission error')}${runLine}\n${esc(clip(d.error))}${foot}` };
|
|
274
|
+
|
|
275
|
+
// --- scheduled missions -------------------------------------------
|
|
276
|
+
case 'schedule_paused': {
|
|
277
|
+
// A paused schedule is the one thing here that stays broken until
|
|
278
|
+
// someone acts, so it says why in plain words. Which toggle carries it
|
|
279
|
+
// follows the cause: the monthly ceiling is a spending guard, repeated
|
|
280
|
+
// failures are a thing only a human can clear. Anything else — a hand
|
|
281
|
+
// on the switch — is news, not a summons.
|
|
282
|
+
const reason = String(d.reason ?? '');
|
|
283
|
+
const gate: keyof NotifyPrefs = reason === 'monthly-cap' ? 'budget' : reason === 'failures' ? 'needsYou' : 'done';
|
|
284
|
+
const why = reason === 'failures'
|
|
285
|
+
? 'two scheduled runs in a row failed'
|
|
286
|
+
: reason === 'monthly-cap'
|
|
287
|
+
? "this month's scheduled spend would go past the ceiling"
|
|
288
|
+
: reason === 'human'
|
|
289
|
+
? 'you paused it'
|
|
290
|
+
: `paused${reason ? ` — ${esc(clip(reason, 60))}` : ''}`;
|
|
291
|
+
return {
|
|
292
|
+
key: `schedule:${d.scheduleId}`, gate,
|
|
293
|
+
// No buttons on purpose: resuming a schedule is a dashboard act, where
|
|
294
|
+
// the cadence, the caps and what it last did are all in view.
|
|
295
|
+
text: `${head('Schedule paused')}\n<b>${esc(clip(d.name, 60))}</b> — ${why}.` +
|
|
296
|
+
`\n<i>Resume it from the Foreman dashboard; there is no resume from here.</i>${foot}`,
|
|
297
|
+
};
|
|
298
|
+
}
|
|
299
|
+
case 'schedule_skipped': {
|
|
300
|
+
// Nothing is wrong and nothing is owed: the cadence simply stepped over
|
|
301
|
+
// a busy project. Keyed by timestamp like a stall, so a run of skips
|
|
302
|
+
// reads as a run of skips rather than one deduped line.
|
|
303
|
+
const reason = String(d.reason ?? '');
|
|
304
|
+
const why = reason === 'project busy'
|
|
305
|
+
? 'the project already had a mission running, so this turn was not started'
|
|
306
|
+
: `not started${reason ? ` — ${esc(clip(reason, 60))}` : ''}`;
|
|
307
|
+
return {
|
|
308
|
+
key: `skip:${d.scheduleId}:${env.ts ?? ''}`, gate: 'done',
|
|
309
|
+
text: `${head('Scheduled run skipped')}\n<b>${esc(clip(d.name, 60))}</b> — ${why}.` +
|
|
310
|
+
`\n<i>The next scheduled run stands.</i>${foot}`,
|
|
311
|
+
};
|
|
312
|
+
}
|
|
269
313
|
default:
|
|
270
314
|
return null;
|
|
271
315
|
}
|
package/src/orchestrator.test.ts
CHANGED
|
@@ -3,17 +3,22 @@ import assert from 'node:assert/strict';
|
|
|
3
3
|
import os from 'node:os';
|
|
4
4
|
import path from 'node:path';
|
|
5
5
|
import { mkdir, mkdtemp, readFile, rm, stat, writeFile } from 'node:fs/promises';
|
|
6
|
+
import { execFileSync } from 'node:child_process';
|
|
6
7
|
import {
|
|
7
8
|
DEFAULT_REPEAT_LIMIT, DIRECTOR_CHARTER, MissionRun, RECENT_LINES, WORKER_CHARTER, WORK_DIR, accumulateUsage,
|
|
8
9
|
activityHint, ensureIgnoreLines, loopingWorkerReport,
|
|
9
10
|
stalledWorkerReport, workerStatusBlock,
|
|
10
11
|
watchRepeats, watchSilence, REPEAT_EXEMPT, observeToolUse,
|
|
11
12
|
DEFAULT_ASK_TIMEOUT_MS, armAskTimeout, unattendedAnswer, unattendedDenyMessage,
|
|
12
|
-
tokenCapLabel, stopReasonOf, budgetWarnDue, doneAtCap,
|
|
13
|
+
tokenCapLabel, stopReasonOf, budgetWarnDue, doneAtCap, renderDeckDiff,
|
|
13
14
|
} from './orchestrator.js';
|
|
14
15
|
import { makePolicy, type PendingPermission } from './policy.js';
|
|
16
|
+
import { REVIEWER_TOOL_POLICY, type CrewPreset } from './crew.js';
|
|
17
|
+
import { captureBaseline } from './deck.js';
|
|
18
|
+
import { changeFingerprint } from './gitwork.js';
|
|
19
|
+
import type { CanUseTool } from '@anthropic-ai/claude-agent-sdk';
|
|
15
20
|
import type { AgentEnv } from './provider.js';
|
|
16
|
-
import type { RunMeta } from './types.js';
|
|
21
|
+
import type { RunMeta, ToolPolicy } from './types.js';
|
|
17
22
|
|
|
18
23
|
function meta(over: Partial<RunMeta> = {}): RunMeta {
|
|
19
24
|
return {
|
|
@@ -1280,3 +1285,512 @@ test('doneAtCap: budget-stopped with every criterion ticked is done, and nothing
|
|
|
1280
1285
|
assert.equal(doneAtCap({ ...budget, usageLimited: true }, []), false, 'a usage limit is not a finish');
|
|
1281
1286
|
assert.equal(doneAtCap({ ...budget, budgetStopped: false }, []), false);
|
|
1282
1287
|
});
|
|
1288
|
+
|
|
1289
|
+
// ---------------------------------------------------------------------------
|
|
1290
|
+
// The reviewer: a read-only worker whose PASS the run cannot be done without
|
|
1291
|
+
// ---------------------------------------------------------------------------
|
|
1292
|
+
|
|
1293
|
+
type ReviewSurface = ToolSurface & {
|
|
1294
|
+
meta: RunMeta;
|
|
1295
|
+
requestReviewTool(a: { presetId: string; notes?: string }): Promise<string>;
|
|
1296
|
+
messageWorkerTool(a: { worker_id: string; message: string }): Promise<string>;
|
|
1297
|
+
crewLine(): string;
|
|
1298
|
+
settleFinalStatus(): Promise<void>;
|
|
1299
|
+
policyFor(agent: string, override?: ToolPolicy): CanUseTool;
|
|
1300
|
+
budgetStopped: boolean;
|
|
1301
|
+
workers: Map<string, { id: string; status: string; sessionId?: string; costUsd: number; overrides?: any; crewPresetId?: string }>;
|
|
1302
|
+
};
|
|
1303
|
+
|
|
1304
|
+
/** The crew a human would pick: one required reviewer, read-only. */
|
|
1305
|
+
function reviewerPreset(over: Partial<CrewPreset> = {}): CrewPreset {
|
|
1306
|
+
return {
|
|
1307
|
+
id: 'reviewer', name: 'Reviewer', kind: 'reviewer', model: 'opus',
|
|
1308
|
+
toolPolicy: 'read-only', requiredForDone: true, brief: 'Review it.',
|
|
1309
|
+
...over,
|
|
1310
|
+
};
|
|
1311
|
+
}
|
|
1312
|
+
|
|
1313
|
+
/**
|
|
1314
|
+
* A run with a crew, whose workers are simulated. `seen` records what
|
|
1315
|
+
* launchWorker passed down, which is how the reviewer's own model and tool
|
|
1316
|
+
* policy are checked without an SDK behind them.
|
|
1317
|
+
*/
|
|
1318
|
+
/**
|
|
1319
|
+
* A folder a review can actually be about: real files, a captured baseline,
|
|
1320
|
+
* and one edit since. Reviews are pinned to what changed, so a run whose
|
|
1321
|
+
* changes cannot be read is a different test (and there is one below).
|
|
1322
|
+
*/
|
|
1323
|
+
async function reviewableFolder(): Promise<string> {
|
|
1324
|
+
const dir = await mkdtemp(path.join(os.tmpdir(), 'crew-'));
|
|
1325
|
+
await writeFile(path.join(dir, 'a.txt'), 'before\n');
|
|
1326
|
+
await captureBaseline(dir, 'run-1');
|
|
1327
|
+
await writeFile(path.join(dir, 'a.txt'), 'after\n');
|
|
1328
|
+
return dir;
|
|
1329
|
+
}
|
|
1330
|
+
|
|
1331
|
+
function crewRun(
|
|
1332
|
+
crew: CrewPreset[],
|
|
1333
|
+
script: (id: string, prompt: string) => Promise<Outcome>,
|
|
1334
|
+
over: Partial<RunMeta> = {},
|
|
1335
|
+
crewEnv?: Record<string, AgentEnv>,
|
|
1336
|
+
) {
|
|
1337
|
+
const events: Array<{ event: string; data: any }> = [];
|
|
1338
|
+
const seen: Array<{ id: string; prompt: string; overrides?: { model?: string; toolPolicy?: ToolPolicy; reason?: string; env?: AgentEnv; nativeCost?: boolean; crewPresetId?: string } }> = [];
|
|
1339
|
+
const run = new MissionRun(
|
|
1340
|
+
meta({ crew, workerModel: 'sonnet', ...over }),
|
|
1341
|
+
(event, data) => events.push({ event, data }), () => {},
|
|
1342
|
+
crewEnv ? { ...noopAgentEnv, crew: crewEnv } : noopAgentEnv,
|
|
1343
|
+
);
|
|
1344
|
+
const t = run as unknown as ReviewSurface;
|
|
1345
|
+
t.runWorker = ((id: string, prompt: string, _resume?: string, overrides?: any) => {
|
|
1346
|
+
seen.push({ id, prompt, overrides });
|
|
1347
|
+
return script(id, prompt);
|
|
1348
|
+
}) as ToolSurface['runWorker'];
|
|
1349
|
+
return { run, t, events, seen };
|
|
1350
|
+
}
|
|
1351
|
+
|
|
1352
|
+
/** The hash of a run whose folder has no baseline: an empty deck, but a stable one. */
|
|
1353
|
+
/** What the gate will hash for a folder: the working tree itself, not the deck. */
|
|
1354
|
+
const fingerprintOf = (folder: string) => changeFingerprint(folder);
|
|
1355
|
+
|
|
1356
|
+
test('the reviewer cannot write, edit or run anything — and no other worker in the run is affected', async () => {
|
|
1357
|
+
const run = new MissionRun(meta({ folder: '/Users/x/proj' }), () => {}, () => {}, noopAgentEnv);
|
|
1358
|
+
const t = run as unknown as ReviewSurface;
|
|
1359
|
+
const opts = { signal: new AbortController().signal, toolUseID: 'u1' } as any;
|
|
1360
|
+
const ordinary = t.policyFor('worker-1');
|
|
1361
|
+
const reviewer = t.policyFor('worker-2', REVIEWER_TOOL_POLICY);
|
|
1362
|
+
|
|
1363
|
+
const inputs: Record<string, Record<string, unknown>> = {
|
|
1364
|
+
Write: { file_path: '/Users/x/proj/a.ts', content: 'x' },
|
|
1365
|
+
Edit: { file_path: '/Users/x/proj/a.ts', old_string: 'a', new_string: 'b' },
|
|
1366
|
+
Bash: { command: 'npm test' },
|
|
1367
|
+
};
|
|
1368
|
+
for (const toolName of ['Write', 'Edit', 'Bash']) {
|
|
1369
|
+
const denied = await reviewer(toolName, inputs[toolName], opts);
|
|
1370
|
+
assert.equal(denied?.behavior, 'deny', `${toolName} must be denied for the reviewer`);
|
|
1371
|
+
const allowed = await ordinary(toolName, inputs[toolName], opts);
|
|
1372
|
+
assert.equal(allowed?.behavior, 'allow', `${toolName} must still be allowed for an ordinary worker`);
|
|
1373
|
+
}
|
|
1374
|
+
// What it does need is untouched: the diff arrives in its brief, but it can
|
|
1375
|
+
// still open the files the diff names.
|
|
1376
|
+
assert.equal((await reviewer('Read', { file_path: '/Users/x/proj/a.ts' }, opts))?.behavior, 'allow');
|
|
1377
|
+
});
|
|
1378
|
+
|
|
1379
|
+
test('request_review refuses a run with no crew, and an unknown preset id names the ones there are', async () => {
|
|
1380
|
+
const { t: none } = crewRun(
|
|
1381
|
+
[], slowWorker(1));
|
|
1382
|
+
none.meta.crew = undefined;
|
|
1383
|
+
assert.match(await none.requestReviewTool({ presetId: 'reviewer' }), /no crew/i);
|
|
1384
|
+
|
|
1385
|
+
const { t, seen } = crewRun(
|
|
1386
|
+
[reviewerPreset(), reviewerPreset({ id: 'security-review', name: 'Security review' })], slowWorker(1));
|
|
1387
|
+
const reply = await t.requestReviewTool({ presetId: 'nobody' });
|
|
1388
|
+
assert.match(reply, /No crew preset "nobody"/);
|
|
1389
|
+
assert.match(reply, /reviewer, security-review/);
|
|
1390
|
+
assert.equal(seen.length, 0, 'a refused review must not spend a worker');
|
|
1391
|
+
});
|
|
1392
|
+
|
|
1393
|
+
test('a PASS is recorded against the current diff, emitted, and pinned to it', async () => {
|
|
1394
|
+
const folder = await reviewableFolder();
|
|
1395
|
+
const { run, t, events, seen } = crewRun(
|
|
1396
|
+
[reviewerPreset()],
|
|
1397
|
+
async () => ({ report: 'I read it all.\nVERDICT: PASS\n- nothing worth blocking on', isError: false }),
|
|
1398
|
+
{ folder },
|
|
1399
|
+
);
|
|
1400
|
+
const reply = await t.requestReviewTool({ presetId: 'reviewer', notes: 'look at the gate' });
|
|
1401
|
+
|
|
1402
|
+
assert.match(reply, /VERDICT: PASS/);
|
|
1403
|
+
assert.match(reply, /nothing worth blocking on/);
|
|
1404
|
+
assert.match(reply, /review last/);
|
|
1405
|
+
|
|
1406
|
+
// The reviewer is an ordinary worker in the worker-N sequence, on its own
|
|
1407
|
+
// model, with the read-only policy attached to it alone.
|
|
1408
|
+
assert.equal(seen[0].id, 'worker-1');
|
|
1409
|
+
assert.equal(seen[0].overrides?.model, 'opus');
|
|
1410
|
+
assert.deepEqual(seen[0].overrides?.toolPolicy, REVIEWER_TOOL_POLICY);
|
|
1411
|
+
assert.match(seen[0].prompt, /Review it\./);
|
|
1412
|
+
assert.match(seen[0].prompt, /look at the gate/, 'the director\'s note reaches the reviewer');
|
|
1413
|
+
|
|
1414
|
+
const v = run.meta.reviews?.[0];
|
|
1415
|
+
assert.ok(v, 'the verdict is recorded on the run');
|
|
1416
|
+
assert.equal(v!.pass, true);
|
|
1417
|
+
assert.equal(v!.presetId, 'reviewer');
|
|
1418
|
+
assert.equal(v!.workerId, 'worker-1');
|
|
1419
|
+
assert.equal(v!.diffHash, await fingerprintOf(run.meta.folder), 'pinned to the work as it stands now');
|
|
1420
|
+
|
|
1421
|
+
const emitted = events.find((e) => e.event === 'review_verdict');
|
|
1422
|
+
assert.ok(emitted, 'review_verdict is emitted');
|
|
1423
|
+
assert.equal(emitted!.data.pass, true);
|
|
1424
|
+
assert.equal(emitted!.data.name, 'Reviewer');
|
|
1425
|
+
assert.equal(emitted!.data.diffHash, await fingerprintOf(run.meta.folder));
|
|
1426
|
+
});
|
|
1427
|
+
|
|
1428
|
+
test('a preset with a provider of its own reviews on that provider; without one it reviews on the workers\'', async () => {
|
|
1429
|
+
const reviewerEnv = { ANTHROPIC_BASE_URL: 'http://127.0.0.1:9911', ANTHROPIC_API_KEY: 'crew' } as unknown as AgentEnv;
|
|
1430
|
+
const pass = async () => ({ report: 'VERDICT: PASS', isError: false });
|
|
1431
|
+
|
|
1432
|
+
const folder = await reviewableFolder();
|
|
1433
|
+
const pinned = crewRun(
|
|
1434
|
+
[reviewerPreset({ providerId: 'kimi' })], pass, { folder }, { reviewer: reviewerEnv },
|
|
1435
|
+
);
|
|
1436
|
+
await pinned.t.requestReviewTool({ presetId: 'reviewer' });
|
|
1437
|
+
assert.equal(pinned.seen[0].overrides?.env, reviewerEnv, 'the reviewer runs on its preset\'s env');
|
|
1438
|
+
|
|
1439
|
+
// No providerId at all: nothing to resolve, so nothing overrides the role.
|
|
1440
|
+
const plain = crewRun([reviewerPreset()], pass, { folder: await reviewableFolder() });
|
|
1441
|
+
await plain.t.requestReviewTool({ presetId: 'reviewer' });
|
|
1442
|
+
assert.equal(plain.seen[0].overrides?.env, undefined, 'no override means the worker env applies');
|
|
1443
|
+
|
|
1444
|
+
// A providerId that dispatch could NOT resolve — deleted between compose and
|
|
1445
|
+
// review — leaves no entry under this preset's id, and must degrade to the
|
|
1446
|
+
// worker env rather than strand the run at its last step.
|
|
1447
|
+
const gone = crewRun(
|
|
1448
|
+
[reviewerPreset({ providerId: 'deleted' })], pass, { folder: await reviewableFolder() }, { 'some-other-preset': reviewerEnv },
|
|
1449
|
+
);
|
|
1450
|
+
const reply = await gone.t.requestReviewTool({ presetId: 'reviewer' });
|
|
1451
|
+
assert.equal(gone.seen[0].overrides?.env, undefined, 'an absent id falls back, it does not fail');
|
|
1452
|
+
assert.match(reply, /VERDICT: PASS/, 'the review still happens');
|
|
1453
|
+
});
|
|
1454
|
+
|
|
1455
|
+
test('a report with no verdict line is recorded as a FAIL, and the director is told so', async () => {
|
|
1456
|
+
const { run, t, events } = crewRun(
|
|
1457
|
+
[reviewerPreset()],
|
|
1458
|
+
async () => ({ report: 'Looks broadly fine to me, I suppose.', isError: false }),
|
|
1459
|
+
{ folder: await reviewableFolder() },
|
|
1460
|
+
);
|
|
1461
|
+
const reply = await t.requestReviewTool({ presetId: 'reviewer' });
|
|
1462
|
+
assert.match(reply, /NO VERDICT/);
|
|
1463
|
+
assert.match(reply, /recorded as a FAIL/i);
|
|
1464
|
+
assert.equal(run.meta.reviews?.[0].pass, false);
|
|
1465
|
+
assert.match(run.meta.reviews![0].findings, /never stated a verdict/);
|
|
1466
|
+
assert.equal(events.find((e) => e.event === 'review_verdict')!.data.pass, false);
|
|
1467
|
+
});
|
|
1468
|
+
|
|
1469
|
+
test('the gate: a required reviewer that never passed keeps the run out of "done"', async () => {
|
|
1470
|
+
const { run, t, events } = crewRun(
|
|
1471
|
+
[reviewerPreset()], slowWorker(1));
|
|
1472
|
+
run.meta.status = 'done';
|
|
1473
|
+
await t.settleFinalStatus();
|
|
1474
|
+
assert.equal(run.meta.status, 'interrupted');
|
|
1475
|
+
const ev = events.find((e) => e.event === 'mission_unreviewed');
|
|
1476
|
+
assert.ok(ev, 'mission_unreviewed says which reviewer and why');
|
|
1477
|
+
assert.equal(ev!.data.blockers[0].reason, 'missing');
|
|
1478
|
+
assert.match(ev!.data.text, /not done/);
|
|
1479
|
+
});
|
|
1480
|
+
|
|
1481
|
+
test('the gate opens on a PASS pinned to the diff the run ends with, and shuts again if it goes stale', async () => {
|
|
1482
|
+
const folder = await reviewableFolder();
|
|
1483
|
+
const { run, t, events } = crewRun([reviewerPreset()], slowWorker(1), { folder });
|
|
1484
|
+
const pinned = await fingerprintOf(folder);
|
|
1485
|
+
assert.ok(pinned, 'the folder has a readable fingerprint');
|
|
1486
|
+
run.meta.reviews = [{
|
|
1487
|
+
presetId: 'reviewer', name: 'Reviewer', pass: true, findings: '', diffHash: pinned,
|
|
1488
|
+
workerId: 'worker-1', at: Date.now(),
|
|
1489
|
+
}];
|
|
1490
|
+
run.meta.status = 'done';
|
|
1491
|
+
await t.settleFinalStatus();
|
|
1492
|
+
assert.equal(run.meta.status, 'done');
|
|
1493
|
+
assert.equal(events.filter((e) => e.event === 'mission_unreviewed').length, 0);
|
|
1494
|
+
|
|
1495
|
+
// The same PASS against a diff that has since moved on is not consent.
|
|
1496
|
+
run.meta.reviews![0].diffHash = 'a-diff-that-no-longer-exists';
|
|
1497
|
+
run.meta.status = 'done';
|
|
1498
|
+
await t.settleFinalStatus();
|
|
1499
|
+
assert.equal(run.meta.status, 'interrupted');
|
|
1500
|
+
assert.equal(events.find((e) => e.event === 'mission_unreviewed')!.data.blockers[0].reason, 'stale');
|
|
1501
|
+
});
|
|
1502
|
+
|
|
1503
|
+
test('a run stopped at its cap with every box ticked is still not done without its required PASS', async () => {
|
|
1504
|
+
const folder = await mkdtemp(path.join(os.tmpdir(), 'foreman-gate-'));
|
|
1505
|
+
await mkdir(path.join(folder, '.foreman'), { recursive: true });
|
|
1506
|
+
await writeFile(path.join(folder, '.foreman', 'MISSION.md'), '# M\n\n## DONE WHEN\n- [x] it works\n');
|
|
1507
|
+
try {
|
|
1508
|
+
for (const crew of [[reviewerPreset()], []]) {
|
|
1509
|
+
const { run, t, events } = crewRun(crew, slowWorker(1), { folder });
|
|
1510
|
+
run.meta.status = 'interrupted';
|
|
1511
|
+
t.budgetStopped = true;
|
|
1512
|
+
await t.settleFinalStatus();
|
|
1513
|
+
if (crew.length) {
|
|
1514
|
+
assert.equal(run.meta.status, 'interrupted', 'no PASS, no promotion at the cap');
|
|
1515
|
+
assert.ok(events.some((e) => e.event === 'mission_unreviewed'));
|
|
1516
|
+
assert.ok(!events.some((e) => e.event === 'mission_done_at_cap'));
|
|
1517
|
+
} else {
|
|
1518
|
+
assert.equal(run.meta.status, 'done', 'without a required reviewer the cap rule is unchanged');
|
|
1519
|
+
assert.ok(events.some((e) => e.event === 'mission_done_at_cap'));
|
|
1520
|
+
}
|
|
1521
|
+
}
|
|
1522
|
+
} finally {
|
|
1523
|
+
await rm(folder, { recursive: true, force: true });
|
|
1524
|
+
}
|
|
1525
|
+
});
|
|
1526
|
+
|
|
1527
|
+
test('renderDeckDiff cuts a large diff and says it did', () => {
|
|
1528
|
+
const file = (p: string, diff: string) => ({ path: p, status: 'modified' as const, additions: 1, deletions: 0, diff });
|
|
1529
|
+
const small = renderDeckDiff({ files: [file('a.ts', '+one')], totals: { files: 1, additions: 1, deletions: 0 } } as any);
|
|
1530
|
+
assert.equal(small.truncated, false);
|
|
1531
|
+
assert.match(small.diff, /--- a\.ts \(modified \+1 -0\)\n\+one/);
|
|
1532
|
+
|
|
1533
|
+
const big = renderDeckDiff({ files: [file('a.ts', 'x'.repeat(200)), file('b.ts', 'y')], totals: { files: 2, additions: 2, deletions: 0 } } as any, 100);
|
|
1534
|
+
assert.equal(big.truncated, true, 'the reviewer must be told it is not seeing everything');
|
|
1535
|
+
// A file the deck itself already cut counts as truncated too.
|
|
1536
|
+
const perFile = renderDeckDiff({
|
|
1537
|
+
files: [{ ...file('a.ts', '+one'), truncated: true }], totals: { files: 1, additions: 1, deletions: 0 },
|
|
1538
|
+
} as any);
|
|
1539
|
+
assert.equal(perFile.truncated, true);
|
|
1540
|
+
});
|
|
1541
|
+
|
|
1542
|
+
test('the charter tells the director to review last, and names the tool', () => {
|
|
1543
|
+
assert.match(DIRECTOR_CHARTER, /request_review/);
|
|
1544
|
+
assert.match(DIRECTOR_CHARTER, /review last/);
|
|
1545
|
+
});
|
|
1546
|
+
|
|
1547
|
+
// --- what the codex review of this branch found, and what now holds -------
|
|
1548
|
+
|
|
1549
|
+
test('the fingerprint is of the work, not of the deck that displays it', async () => {
|
|
1550
|
+
const dir = await mkdtemp(path.join(os.tmpdir(), 'fingerprint-'));
|
|
1551
|
+
// The deck stops at 200 files and carries no binary content, so a hash
|
|
1552
|
+
// derived from it would be blind to both of these.
|
|
1553
|
+
for (let i = 0; i < 250; i++) await writeFile(path.join(dir, `f${i}.txt`), `file ${i}\n`);
|
|
1554
|
+
await writeFile(path.join(dir, 'image.bin'), Buffer.from([0, 1, 2, 3]));
|
|
1555
|
+
const first = await changeFingerprint(dir);
|
|
1556
|
+
assert.ok(first);
|
|
1557
|
+
|
|
1558
|
+
await writeFile(path.join(dir, 'f240.txt'), 'changed well past the deck cap\n');
|
|
1559
|
+
const afterLateFile = await changeFingerprint(dir);
|
|
1560
|
+
assert.notEqual(afterLateFile, first, 'a change beyond the deck cap moves the fingerprint');
|
|
1561
|
+
|
|
1562
|
+
await writeFile(path.join(dir, 'image.bin'), Buffer.from([9, 9, 9, 9]));
|
|
1563
|
+
const afterBinary = await changeFingerprint(dir);
|
|
1564
|
+
assert.notEqual(afterBinary, afterLateFile, 'and so does a swapped binary of the same size');
|
|
1565
|
+
|
|
1566
|
+
// Foreman's own directory is not the work under review.
|
|
1567
|
+
await mkdir(path.join(dir, '.foreman'), { recursive: true });
|
|
1568
|
+
await writeFile(path.join(dir, '.foreman', 'MISSION.md'), 'churn\n');
|
|
1569
|
+
assert.equal(await changeFingerprint(dir), afterBinary, '.foreman does not move it');
|
|
1570
|
+
});
|
|
1571
|
+
|
|
1572
|
+
test('a run whose changes cannot be shown gets no review, rather than a review of nothing', async () => {
|
|
1573
|
+
// No baseline captured: deckFor answers with an empty deck instead of
|
|
1574
|
+
// throwing, which used to look exactly like "nothing changed".
|
|
1575
|
+
const folder = await mkdtemp(path.join(os.tmpdir(), 'nobaseline-'));
|
|
1576
|
+
await writeFile(path.join(folder, 'a.txt'), 'work happened here\n');
|
|
1577
|
+
const { t, seen } = crewRun([reviewerPreset()], async () => ({ report: 'VERDICT: PASS', isError: false }), { folder });
|
|
1578
|
+
const reply = await t.requestReviewTool({ presetId: 'reviewer' });
|
|
1579
|
+
assert.match(reply, /no baseline/i);
|
|
1580
|
+
assert.equal(seen.length, 0, 'and no worker is spent reviewing an empty diff');
|
|
1581
|
+
});
|
|
1582
|
+
|
|
1583
|
+
test('a reviewer stays read-only when the director follows up on it', async () => {
|
|
1584
|
+
const folder = await reviewableFolder();
|
|
1585
|
+
const { t, seen } = crewRun([reviewerPreset()], async () => ({ report: 'VERDICT: FAIL\n- one thing', isError: false }), { folder });
|
|
1586
|
+
await t.requestReviewTool({ presetId: 'reviewer' });
|
|
1587
|
+
const first = seen[0].overrides;
|
|
1588
|
+
assert.ok(first?.toolPolicy, 'the review ran under the reviewer policy');
|
|
1589
|
+
|
|
1590
|
+
// message_worker resumes the same agent; without its overrides it would come
|
|
1591
|
+
// back as an ordinary worker that can write, wearing the reviewer's id.
|
|
1592
|
+
const w = t.workers.get('worker-1')!;
|
|
1593
|
+
w.sessionId = 'session-1';
|
|
1594
|
+
w.status = 'done';
|
|
1595
|
+
await t.messageWorkerTool({ worker_id: 'worker-1', message: 'what exactly did you mean?' });
|
|
1596
|
+
assert.deepEqual(seen[1].overrides?.toolPolicy, first!.toolPolicy, 'the follow-up keeps the read-only policy');
|
|
1597
|
+
assert.equal(seen[1].overrides?.model, first!.model, 'and the preset\'s model');
|
|
1598
|
+
});
|
|
1599
|
+
|
|
1600
|
+
test('a reviewer on its own priced provider has its dollars counted', async () => {
|
|
1601
|
+
const reviewerEnv = { ANTHROPIC_BASE_URL: 'http://127.0.0.1:9911' } as unknown as AgentEnv;
|
|
1602
|
+
const folder = await reviewableFolder();
|
|
1603
|
+
const pass = async () => ({ report: 'VERDICT: PASS', isError: false });
|
|
1604
|
+
|
|
1605
|
+
// Workers behind a free gateway, reviewer on a paid provider of its own: its
|
|
1606
|
+
// bill arrives whatever the workers cost, so the cap has to see it.
|
|
1607
|
+
const priced = new MissionRun(
|
|
1608
|
+
meta({ crew: [reviewerPreset({ providerId: 'openai' })], workerModel: 'sonnet', folder }),
|
|
1609
|
+
() => {}, () => {},
|
|
1610
|
+
{ ...noopAgentEnv, crew: { reviewer: reviewerEnv } },
|
|
1611
|
+
{}, undefined, { director: 'free', worker: 'free', crew: { reviewer: { basis: 'priced', native: true } } },
|
|
1612
|
+
);
|
|
1613
|
+
const seen: any[] = [];
|
|
1614
|
+
(priced as any).runWorker = (id: string, prompt: string, _r?: string, overrides?: any) => {
|
|
1615
|
+
seen.push(overrides); return pass();
|
|
1616
|
+
};
|
|
1617
|
+
await (priced as unknown as ReviewSurface).requestReviewTool({ presetId: 'reviewer' });
|
|
1618
|
+
assert.equal(seen[0].nativeCost, true, 'the reviewer\'s spend is counted as real money');
|
|
1619
|
+
|
|
1620
|
+
// The same preset on a free provider is not suddenly priced.
|
|
1621
|
+
const free = new MissionRun(
|
|
1622
|
+
meta({ crew: [reviewerPreset({ providerId: 'ollama' })], workerModel: 'sonnet', folder }),
|
|
1623
|
+
() => {}, () => {},
|
|
1624
|
+
{ ...noopAgentEnv, crew: { reviewer: reviewerEnv } },
|
|
1625
|
+
{}, undefined, { director: 'free', worker: 'free', crew: { reviewer: { basis: 'free', native: true } } },
|
|
1626
|
+
);
|
|
1627
|
+
const seenFree: any[] = [];
|
|
1628
|
+
(free as any).runWorker = (id: string, prompt: string, _r?: string, overrides?: any) => {
|
|
1629
|
+
seenFree.push(overrides); return pass();
|
|
1630
|
+
};
|
|
1631
|
+
await (free as unknown as ReviewSurface).requestReviewTool({ presetId: 'reviewer' });
|
|
1632
|
+
assert.equal(seenFree[0].nativeCost, false);
|
|
1633
|
+
});
|
|
1634
|
+
|
|
1635
|
+
test('the director is told which reviews it must get, by id', () => {
|
|
1636
|
+
const run = new MissionRun(
|
|
1637
|
+
meta({ crew: [reviewerPreset(), reviewerPreset({ id: 'security-review', name: 'Security review', requiredForDone: false })] }),
|
|
1638
|
+
() => {}, () => {}, noopAgentEnv,
|
|
1639
|
+
);
|
|
1640
|
+
const line = (run as unknown as ReviewSurface).crewLine();
|
|
1641
|
+
assert.match(line, /reviewer — Reviewer \(opus\): REQUIRED/);
|
|
1642
|
+
assert.match(line, /security-review — Security review \(opus\): optional/);
|
|
1643
|
+
assert.match(line, /request_review/, 'and how to call for one');
|
|
1644
|
+
assert.match(line, /stale/, 'and that changing things afterwards undoes it');
|
|
1645
|
+
|
|
1646
|
+
// A run with no crew says nothing at all, rather than an empty heading.
|
|
1647
|
+
const bare = new MissionRun(meta(), () => {}, () => {}, noopAgentEnv);
|
|
1648
|
+
assert.equal((bare as unknown as ReviewSurface).crewLine(), '');
|
|
1649
|
+
});
|
|
1650
|
+
|
|
1651
|
+
test('a commit is a change: a PASS does not survive one', async () => {
|
|
1652
|
+
const dir = await mkdtemp(path.join(os.tmpdir(), 'committed-'));
|
|
1653
|
+
const git = (...args: string[]) => execFileSync('git', args, { cwd: dir, stdio: 'pipe', env: { ...process.env, GIT_CONFIG_GLOBAL: '/dev/null' } });
|
|
1654
|
+
git('init', '-q', '-b', 'main');
|
|
1655
|
+
git('config', 'user.email', 'me@example.com');
|
|
1656
|
+
git('config', 'user.name', 'Me');
|
|
1657
|
+
await writeFile(path.join(dir, 'a.txt'), 'first\n');
|
|
1658
|
+
git('add', '-A'); git('commit', '-q', '-m', 'one');
|
|
1659
|
+
|
|
1660
|
+
const reviewed = await changeFingerprint(dir);
|
|
1661
|
+
assert.ok(reviewed);
|
|
1662
|
+
|
|
1663
|
+
// The director keeps working and commits. `git diff HEAD` is empty again —
|
|
1664
|
+
// which is exactly how a stale PASS used to look current.
|
|
1665
|
+
await writeFile(path.join(dir, 'a.txt'), 'second\n');
|
|
1666
|
+
git('add', '-A'); git('commit', '-q', '-m', 'two');
|
|
1667
|
+
assert.notEqual(await changeFingerprint(dir), reviewed, 'the committed change moves the fingerprint');
|
|
1668
|
+
});
|
|
1669
|
+
|
|
1670
|
+
test('a reviewer resumed after a restart is rebuilt from the crew, not as a plain worker', async () => {
|
|
1671
|
+
const folder = await reviewableFolder();
|
|
1672
|
+
// A run that came back from disk: the worker record survived, the live
|
|
1673
|
+
// overrides (which hold a credential) did not.
|
|
1674
|
+
const { t, seen } = crewRun([reviewerPreset()], async () => ({ report: 'VERDICT: PASS', isError: false }), {
|
|
1675
|
+
folder,
|
|
1676
|
+
workers: [{ id: 'worker-1', status: 'done', costUsd: 0.1, task: 'review', sessionId: 'session-1', crewPresetId: 'reviewer' }],
|
|
1677
|
+
});
|
|
1678
|
+
await t.messageWorkerTool({ worker_id: 'worker-1', message: 'one more question' });
|
|
1679
|
+
assert.ok(seen[0].overrides?.toolPolicy, 'the rebuilt launch is still read-only');
|
|
1680
|
+
assert.equal(seen[0].overrides?.model, 'opus', 'and still on the preset\'s model');
|
|
1681
|
+
assert.equal(seen[0].overrides?.crewPresetId, 'reviewer');
|
|
1682
|
+
});
|
|
1683
|
+
|
|
1684
|
+
test('a paid reviewer makes the run priced before it runs, so the cap is live', async () => {
|
|
1685
|
+
const folder = await reviewableFolder();
|
|
1686
|
+
const events: Array<{ event: string; data: any }> = [];
|
|
1687
|
+
const run = new MissionRun(
|
|
1688
|
+
meta({ crew: [reviewerPreset({ providerId: 'openai' })], workerModel: 'sonnet', folder, costBasis: 'free' }),
|
|
1689
|
+
(event, data) => events.push({ event, data }), () => {},
|
|
1690
|
+
{ ...noopAgentEnv, crew: { reviewer: {} as AgentEnv } },
|
|
1691
|
+
{}, undefined, { director: 'free', worker: 'free', crew: { reviewer: { basis: 'priced', native: true } } },
|
|
1692
|
+
);
|
|
1693
|
+
(run as any).runWorker = async () => ({ report: 'VERDICT: PASS', isError: false });
|
|
1694
|
+
await (run as unknown as ReviewSurface).requestReviewTool({ presetId: 'reviewer' });
|
|
1695
|
+
|
|
1696
|
+
assert.equal(run.meta.costBasis, 'priced', 'the run is priced from the moment the paid reviewer starts');
|
|
1697
|
+
assert.equal(run.meta.metered, true, 'so enforceBudget and capReached stop standing down');
|
|
1698
|
+
const said = events.find((e) => e.event === 'settings_changed');
|
|
1699
|
+
assert.match(said!.data.changes[0], /dollar cap is live/, 'and the change is announced, not silent');
|
|
1700
|
+
});
|
|
1701
|
+
|
|
1702
|
+
test('a reviewer on its own paid provider is charged once, not twice', async () => {
|
|
1703
|
+
const folder = await reviewableFolder();
|
|
1704
|
+
// Workers with a published rate card AND a reviewer billing in dollars: the
|
|
1705
|
+
// tokens used to be priced at the worker's rates and the SDK's dollars added
|
|
1706
|
+
// on top, charging the same review twice and stopping runs early.
|
|
1707
|
+
const run = new MissionRun(
|
|
1708
|
+
meta({ crew: [reviewerPreset({ providerId: 'openai' })], workerModel: 'sonnet', folder, budgetUsd: 5 }),
|
|
1709
|
+
() => {}, () => {},
|
|
1710
|
+
{ ...noopAgentEnv, crew: { reviewer: {} as AgentEnv } },
|
|
1711
|
+
{ worker: { inputPerMTok: 1000, outputPerMTok: 1000 } as any },
|
|
1712
|
+
undefined,
|
|
1713
|
+
{ director: 'priced', worker: 'priced', crew: { reviewer: { basis: 'priced', native: true } } },
|
|
1714
|
+
);
|
|
1715
|
+
(run as any).runWorker = async (_id: string, _p: string, _r: string | undefined, overrides: any) => {
|
|
1716
|
+
(run as any).addUsage({ input_tokens: 1_000_000, output_tokens: 1_000_000 }, overrides?.priceRole ?? 'worker', overrides?.nativeCost === true);
|
|
1717
|
+
(run as any).addCost(0.25, overrides?.priceRole ?? 'worker', overrides?.nativeCost === true);
|
|
1718
|
+
return { report: 'VERDICT: PASS', isError: false };
|
|
1719
|
+
};
|
|
1720
|
+
await (run as unknown as ReviewSurface).requestReviewTool({ presetId: 'reviewer' });
|
|
1721
|
+
assert.equal(run.meta.costUsd, 0.25, 'the provider\'s own figure, and only that');
|
|
1722
|
+
});
|
|
1723
|
+
|
|
1724
|
+
test('a priced gateway reviewer is not treated as a native bill, and never drives the browser', async () => {
|
|
1725
|
+
const folder = await reviewableFolder();
|
|
1726
|
+
// Priced, but through a gateway: the SDK's dollar figure is not the bill,
|
|
1727
|
+
// the ledger's is. Counting both is how a review gets charged twice.
|
|
1728
|
+
const gateway = new MissionRun(
|
|
1729
|
+
meta({ crew: [reviewerPreset({ providerId: 'kimi' })], workerModel: 'sonnet', folder, browserTools: true, costBasis: 'free' }),
|
|
1730
|
+
() => {}, () => {},
|
|
1731
|
+
{ ...noopAgentEnv, crew: { reviewer: {} as AgentEnv } },
|
|
1732
|
+
{}, undefined, { director: 'free', worker: 'free', crew: { reviewer: { basis: 'priced', native: false } } },
|
|
1733
|
+
);
|
|
1734
|
+
const seen: any[] = [];
|
|
1735
|
+
(gateway as any).runWorker = (_i: string, _p: string, _r: string | undefined, o: any) => {
|
|
1736
|
+
seen.push(o); return Promise.resolve({ report: 'VERDICT: PASS', isError: false });
|
|
1737
|
+
};
|
|
1738
|
+
await (gateway as unknown as ReviewSurface).requestReviewTool({ presetId: 'reviewer' });
|
|
1739
|
+
assert.equal(seen[0].nativeCost, false, 'a gateway bill is not the SDK\'s figure');
|
|
1740
|
+
assert.equal(gateway.meta.costBasis, 'priced', 'but it is still real money, so the cap is live');
|
|
1741
|
+
assert.equal(seen[0].noBrowser, true, 'and a read-only reviewer is given no browser to click with');
|
|
1742
|
+
|
|
1743
|
+
// A specialist that is meant to run things keeps the browser.
|
|
1744
|
+
const specialist = new MissionRun(
|
|
1745
|
+
meta({ crew: [reviewerPreset({ id: 'runner', toolPolicy: 'default' })], workerModel: 'sonnet', folder, browserTools: true }),
|
|
1746
|
+
() => {}, () => {}, noopAgentEnv,
|
|
1747
|
+
);
|
|
1748
|
+
const seenTwo: any[] = [];
|
|
1749
|
+
(specialist as any).runWorker = (_i: string, _p: string, _r: string | undefined, o: any) => {
|
|
1750
|
+
seenTwo.push(o); return Promise.resolve({ report: 'VERDICT: PASS', isError: false });
|
|
1751
|
+
};
|
|
1752
|
+
await (specialist as unknown as ReviewSurface).requestReviewTool({ presetId: 'runner' });
|
|
1753
|
+
assert.equal(seenTwo[0].noBrowser, false);
|
|
1754
|
+
});
|
|
1755
|
+
|
|
1756
|
+
test('a gateway reviewer is not priced with the workers\' rate card, and keeps its limits on a retry', async () => {
|
|
1757
|
+
const folder = await reviewableFolder();
|
|
1758
|
+
const run = new MissionRun(
|
|
1759
|
+
meta({ crew: [reviewerPreset({ providerId: 'kimi' })], workerModel: 'sonnet', folder, browserTools: true }),
|
|
1760
|
+
() => {}, () => {},
|
|
1761
|
+
{ ...noopAgentEnv, crew: { reviewer: {} as AgentEnv } },
|
|
1762
|
+
{ worker: { inputPerMTok: 1000, outputPerMTok: 1000 } as any },
|
|
1763
|
+
undefined,
|
|
1764
|
+
{ director: 'priced', worker: 'priced', crew: { reviewer: { basis: 'priced', native: false } } },
|
|
1765
|
+
);
|
|
1766
|
+
const seen: any[] = [];
|
|
1767
|
+
(run as any).runWorker = (_i: string, _p: string, _r: string | undefined, o: any) => {
|
|
1768
|
+
seen.push(o);
|
|
1769
|
+
(run as any).addUsage({ input_tokens: 1_000_000, output_tokens: 1_000_000 }, o?.priceRole ?? 'worker',
|
|
1770
|
+
o?.nativeCost === true || o?.ownProvider === true);
|
|
1771
|
+
return Promise.resolve({ report: 'VERDICT: FAIL\n- no', isError: false });
|
|
1772
|
+
};
|
|
1773
|
+
await (run as unknown as ReviewSurface).requestReviewTool({ presetId: 'reviewer' });
|
|
1774
|
+
assert.equal(seen[0].ownProvider, true);
|
|
1775
|
+
assert.equal(run.meta.costUsd, 0, 'the workers\' rates do not describe another endpoint\'s tokens');
|
|
1776
|
+
assert.ok((run.meta.usage?.inputTokens ?? 0) > 0, 'but the tokens are still counted');
|
|
1777
|
+
});
|
|
1778
|
+
|
|
1779
|
+
test('an unreadable directory is not an empty one', async () => {
|
|
1780
|
+
const dir = await mkdtemp(path.join(os.tmpdir(), 'unreadable-'));
|
|
1781
|
+
await writeFile(path.join(dir, 'visible.txt'), 'hello\n');
|
|
1782
|
+
const closed = path.join(dir, 'closed');
|
|
1783
|
+
await mkdir(closed, { recursive: true });
|
|
1784
|
+
await writeFile(path.join(closed, 'secret.txt'), 'work\n');
|
|
1785
|
+
const before = await changeFingerprint(dir);
|
|
1786
|
+
assert.ok(before);
|
|
1787
|
+
|
|
1788
|
+
// chmod 000: the walk cannot see inside, and must say so rather than
|
|
1789
|
+
// hashing the part of the tree it managed to read.
|
|
1790
|
+
execFileSync('chmod', ['000', closed]);
|
|
1791
|
+
try {
|
|
1792
|
+
assert.equal(await changeFingerprint(dir), null, 'cannot verify, so no fingerprint');
|
|
1793
|
+
} finally {
|
|
1794
|
+
execFileSync('chmod', ['755', closed]);
|
|
1795
|
+
}
|
|
1796
|
+
});
|