@amenophis1er/foreman 0.1.16 → 0.1.18
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +74 -3
- package/package.json +1 -1
- package/src/crew.test.ts +376 -0
- package/src/crew.ts +330 -0
- package/src/fleet-planner.test.ts +39 -1
- package/src/fleet-planner.ts +83 -3
- package/src/gitwork.test.ts +133 -2
- package/src/gitwork.ts +211 -2
- package/src/mcp.test.ts +38 -1
- package/src/mcp.ts +86 -5
- package/src/notify/commands.test.ts +2 -0
- package/src/notify/commands.ts +6 -0
- package/src/notify/telegram.ts +1 -0
- package/src/notify.test.ts +61 -0
- package/src/notify.ts +45 -1
- package/src/orchestrator.test.ts +516 -2
- package/src/orchestrator.ts +514 -69
- package/src/run-crew.test.ts +99 -0
- package/src/run-crew.ts +101 -0
- package/src/schedule-guards.test.ts +236 -0
- package/src/schedule-guards.ts +149 -0
- package/src/schedule.test.ts +240 -0
- package/src/schedule.ts +343 -0
- package/src/server.ts +674 -11
- package/src/store.test.ts +81 -1
- package/src/store.ts +117 -3
- package/src/types.ts +92 -0
- package/ui/dist/assets/index-0QuGXbFg.js +76 -0
- package/ui/dist/index.html +1 -1
- package/ui/dist/assets/index-DOVnExqF.js +0 -68
package/src/orchestrator.ts
CHANGED
|
@@ -3,8 +3,8 @@
|
|
|
3
3
|
*
|
|
4
4
|
* Responsibilities:
|
|
5
5
|
* - Spawn the director session with its charter and in-process MCP tools
|
|
6
|
-
* (spawn_worker /
|
|
7
|
-
* ask_human).
|
|
6
|
+
* (spawn_worker / request_review / check_workers / wait_for_worker /
|
|
7
|
+
* message_worker / ask_human).
|
|
8
8
|
* - Run workers as separate, resumable SDK sessions, concurrently with the
|
|
9
9
|
* director's own turn: spawn_worker returns as soon as the worker exists,
|
|
10
10
|
* and the director reads progress and results back through check_workers
|
|
@@ -95,9 +95,14 @@ import type { AgentEnv } from './provider.js';
|
|
|
95
95
|
import { generateRunTitle } from './title.js';
|
|
96
96
|
import { combineBasis, costBasisOf, isPriced, type CostBasis } from './types.js';
|
|
97
97
|
import { priceUsage, type ModelPrice } from './prices.js';
|
|
98
|
-
import { captureBaseline } from './deck.js';
|
|
98
|
+
import { captureBaseline, deckFor, type Deck } from './deck.js';
|
|
99
|
+
import { changeFingerprint } from './gitwork.js';
|
|
99
100
|
import { memorySection, readMemory, writeMemory } from './memory.js';
|
|
100
|
-
import
|
|
101
|
+
import {
|
|
102
|
+
REVIEWER_TOOL_POLICY, diffHash, parseVerdict, reviewBlockers, reviewBriefFor, unreviewedText,
|
|
103
|
+
type ReviewBlocker, type ReviewVerdict,
|
|
104
|
+
} from './crew.js';
|
|
105
|
+
import type { RunMeta, ToolPolicy, TokenUsage, WorkerMeta, WorkerProgress } from './types.js';
|
|
101
106
|
|
|
102
107
|
/** A run's usage before its first `result` message. */
|
|
103
108
|
function emptyUsage(): TokenUsage {
|
|
@@ -291,6 +296,64 @@ export function doneAtCap(
|
|
|
291
296
|
return Array.isArray(unmet) && unmet.length === 0;
|
|
292
297
|
}
|
|
293
298
|
|
|
299
|
+
/**
|
|
300
|
+
* The lines under the mission doc's DONE WHEN heading, to the next heading —
|
|
301
|
+
* or null when the doc has no such section.
|
|
302
|
+
*
|
|
303
|
+
* One parse, two readers: the end-of-run check that counts unticked boxes, and
|
|
304
|
+
* the reviewer's brief, which quotes the criteria verbatim so the reviewer
|
|
305
|
+
* judges the diff against the same contract Foreman judges the run against.
|
|
306
|
+
* Two parsers would eventually disagree about what "the criteria" are, and the
|
|
307
|
+
* one place that must not happen is the gate.
|
|
308
|
+
*/
|
|
309
|
+
export function doneWhenSection(doc: string): string[] | null {
|
|
310
|
+
const lines = doc.split('\n');
|
|
311
|
+
const start = lines.findIndex((l) => /^#{1,6}\s*DONE\s*WHEN/i.test(l.trim()));
|
|
312
|
+
if (start === -1) return null;
|
|
313
|
+
const out: string[] = [];
|
|
314
|
+
for (const line of lines.slice(start + 1)) {
|
|
315
|
+
// The section ends at the next heading; checkboxes below it are the plan.
|
|
316
|
+
if (/^#{1,6}\s/.test(line)) break;
|
|
317
|
+
out.push(line);
|
|
318
|
+
}
|
|
319
|
+
return out;
|
|
320
|
+
}
|
|
321
|
+
|
|
322
|
+
/**
|
|
323
|
+
* How much of the run's diff fits in a reviewer's brief. The deck is already
|
|
324
|
+
* capped per file and per file count; this is the cap on the whole thing, so a
|
|
325
|
+
* mission that touched two hundred files does not hand its reviewer a prompt
|
|
326
|
+
* nothing will read to the end of. When it bites, the brief says so and the
|
|
327
|
+
* reviewer is told to open the files itself — it has Read, Glob and Grep.
|
|
328
|
+
*/
|
|
329
|
+
export const REVIEW_DIFF_MAX_CHARS = 120_000;
|
|
330
|
+
|
|
331
|
+
/**
|
|
332
|
+
* A deck rendered as one diff text for the reviewer, with a header line per
|
|
333
|
+
* file so a truncated body is still attributable. `truncated` is true when
|
|
334
|
+
* this cut anything OR when the deck had already cut a file's own diff —
|
|
335
|
+
* either way the reviewer is looking at less than the whole change and must
|
|
336
|
+
* be told, because a reviewer that believes it saw everything passes on what
|
|
337
|
+
* it did not see.
|
|
338
|
+
*/
|
|
339
|
+
export function renderDeckDiff(deck: Deck, max = REVIEW_DIFF_MAX_CHARS): { diff: string; truncated: boolean } {
|
|
340
|
+
const files = deck.files ?? [];
|
|
341
|
+
let truncated = files.length < (deck.totals?.files ?? files.length);
|
|
342
|
+
const parts: string[] = [];
|
|
343
|
+
let used = 0;
|
|
344
|
+
for (const f of files) {
|
|
345
|
+
if (f.truncated) truncated = true;
|
|
346
|
+
const head = `--- ${f.path} (${f.status} +${f.additions} -${f.deletions})`;
|
|
347
|
+
const body = f.binary ? '(binary file)' : (f.diff ?? '(no diff available)');
|
|
348
|
+
const block = `${head}\n${body}`;
|
|
349
|
+
if (used + block.length > max) { truncated = true; break; }
|
|
350
|
+
parts.push(block);
|
|
351
|
+
used += block.length + 1;
|
|
352
|
+
}
|
|
353
|
+
const diff = parts.join('\n') || (files.length ? '' : '(this run changed no files)');
|
|
354
|
+
return { diff, truncated };
|
|
355
|
+
}
|
|
356
|
+
|
|
294
357
|
export function stopReasonOf(cap: string): 'budget' | 'turns' | 'time' | 'tokens' {
|
|
295
358
|
if (cap.startsWith('TURN')) return 'turns';
|
|
296
359
|
if (cap.startsWith('TIME')) return 'time';
|
|
@@ -631,9 +694,32 @@ type AgentRole = 'director' | 'worker';
|
|
|
631
694
|
interface WorkerOverrides {
|
|
632
695
|
/** Set on the one continuation a worker gets after the turn cap. */
|
|
633
696
|
continued?: boolean;
|
|
697
|
+
/** The crew preset this worker is, recorded on its meta so a resume can rebuild this. */
|
|
698
|
+
crewPresetId?: string;
|
|
699
|
+
/** Withhold the browser from this worker, whatever the run allows. */
|
|
700
|
+
noBrowser?: boolean;
|
|
701
|
+
/** This agent runs on a provider of its own, so no role's rate card applies to it. */
|
|
702
|
+
ownProvider?: boolean;
|
|
703
|
+
/**
|
|
704
|
+
* Count this agent's dollars as real spend whatever the role it is priced
|
|
705
|
+
* under would normally do. Set for a crew preset on its own priced
|
|
706
|
+
* provider: its bill arrives regardless of how the run's workers are
|
|
707
|
+
* served, and a cap that cannot see it is not a cap.
|
|
708
|
+
*/
|
|
709
|
+
nativeCost?: boolean;
|
|
634
710
|
env?: AgentEnv;
|
|
635
711
|
model?: string;
|
|
636
712
|
priceRole?: AgentRole;
|
|
713
|
+
/**
|
|
714
|
+
* Tool rules for THIS worker only, merged over the run's own policy.
|
|
715
|
+
*
|
|
716
|
+
* A reviewer is a worker that must not be able to change what it is
|
|
717
|
+
* judging, and the run's policy cannot say that — it applies to the whole
|
|
718
|
+
* crew, and tightening it for everyone would stop the workers doing the
|
|
719
|
+
* work. So the restriction travels with the one agent it is about; every
|
|
720
|
+
* other worker and the director are built from the run's policy unchanged.
|
|
721
|
+
*/
|
|
722
|
+
toolPolicy?: ToolPolicy;
|
|
637
723
|
/** For the transcript and the report: why this worker exists. */
|
|
638
724
|
reason?: string;
|
|
639
725
|
}
|
|
@@ -738,25 +824,31 @@ directing worker agents. Non-negotiable rules, in priority order:
|
|
|
738
824
|
after that the task comes back to you half-done. So give a worker a task it
|
|
739
825
|
can finish in that many steps — split the big ones — rather than one brief
|
|
740
826
|
that has to be rescued twice.
|
|
741
|
-
4.
|
|
827
|
+
4. GET THE REVIEW LAST. If this run has a crew reviewer marked required, call
|
|
828
|
+
mcp__foreman__request_review with its preset id AFTER your final change and
|
|
829
|
+
before you tick the last box. It reads the diff and answers PASS or FAIL;
|
|
830
|
+
Foreman will not record the run as done without that PASS, and a change made
|
|
831
|
+
after a PASS makes it stale — the reviewer would have passed code that no
|
|
832
|
+
longer exists — so review last, then stop.
|
|
833
|
+
5. REPORT WHAT YOU SEE. Judge the work as a competent professional would, not
|
|
742
834
|
only against the letter of the acceptance criteria. If you observe a defect
|
|
743
835
|
the criteria did not name — tap targets too small to use, unreadable
|
|
744
836
|
contrast, a broken layout, a hazard, an obviously wrong result — fix it
|
|
745
837
|
when it is clearly in scope, and otherwise say so plainly in your final
|
|
746
838
|
summary and in MISSION.md. Staying silent about a problem you could see is
|
|
747
839
|
a failed mission even when every listed box is ticked.
|
|
748
|
-
|
|
840
|
+
6. DECIDE AND RECORD, DON'T ASK. You are running unattended more often than
|
|
749
841
|
not. Make the reasonable call, write it and the reasoning into MISSION.md,
|
|
750
842
|
and continue. Reserve mcp__foreman__ask_human for decisions that are
|
|
751
843
|
irreversible or that spend money the mission was not given — those you ask
|
|
752
844
|
and wait for. A question left unanswered for ${DEFAULT_ASK_TIMEOUT_MS / 60_000}
|
|
753
845
|
minutes is auto-answered "decide yourself"; treat that answer as the
|
|
754
846
|
human's, record what you decided, and do not ask it again.
|
|
755
|
-
|
|
847
|
+
7. NEVER modify Foreman itself, its server, or any oversight tooling. Tooling
|
|
756
848
|
failure is an escalation, never a self-repair.
|
|
757
|
-
|
|
849
|
+
8. When DONE WHEN is verified, update MISSION.md (all boxes ticked, final log
|
|
758
850
|
entry) and end with a short summary of what was built and how you verified it.
|
|
759
|
-
|
|
851
|
+
9. LEAVE NOTES FOR THE NEXT CREW. Before you finish, call mcp__foreman__remember
|
|
760
852
|
with the whole project memory as it should read now: how to run and test
|
|
761
853
|
the project, ports and paths that matter, conventions and the reasons
|
|
762
854
|
behind them, traps you fell into. Facts, one line each, a page at most.
|
|
@@ -805,6 +897,13 @@ const WORKER_CONTINUE_PROMPT =
|
|
|
805
897
|
*/
|
|
806
898
|
interface WorkerRuntime extends WorkerMeta {
|
|
807
899
|
q?: Query;
|
|
900
|
+
/**
|
|
901
|
+
* The overrides this worker was launched with, kept so a follow-up message
|
|
902
|
+
* resumes the same agent rather than an ordinary worker wearing its id — a
|
|
903
|
+
* reviewer's read-only policy, model and provider must outlive its first
|
|
904
|
+
* turn.
|
|
905
|
+
*/
|
|
906
|
+
overrides?: WorkerOverrides;
|
|
808
907
|
promise?: Promise<WorkerOutcome>;
|
|
809
908
|
done?: Promise<void>;
|
|
810
909
|
settle?: () => void;
|
|
@@ -812,7 +911,7 @@ interface WorkerRuntime extends WorkerMeta {
|
|
|
812
911
|
reportShown?: boolean;
|
|
813
912
|
}
|
|
814
913
|
|
|
815
|
-
const RUNTIME_ONLY: ReadonlyArray<keyof WorkerRuntime> = ['q', 'promise', 'done', 'settle', 'reportShown'];
|
|
914
|
+
const RUNTIME_ONLY: ReadonlyArray<keyof WorkerRuntime> = ['q', 'promise', 'done', 'settle', 'reportShown', 'overrides'];
|
|
816
915
|
|
|
817
916
|
export class MissionRun {
|
|
818
917
|
readonly meta: RunMeta;
|
|
@@ -882,8 +981,15 @@ export class MissionRun {
|
|
|
882
981
|
* provider model, and it is decided by which model each role was given.
|
|
883
982
|
* Passed in rather than derived here so exactly one module decides what an
|
|
884
983
|
* agent can authenticate as; see provider.ts.
|
|
984
|
+
*
|
|
985
|
+
* `crew` is the same thing for a crew preset that named a provider of its
|
|
986
|
+
* own, keyed by preset id, and it is deliberately partial: a preset with no
|
|
987
|
+
* `providerId`, or one whose provider no longer resolves, simply has no
|
|
988
|
+
* entry and runs on the worker env. Optional on the argument rather than a
|
|
989
|
+
* parameter of its own so every existing call site — and every test that
|
|
990
|
+
* constructs a run with two roles — keeps working untouched.
|
|
885
991
|
*/
|
|
886
|
-
private readonly agentEnv: { director: AgentEnv; worker: AgentEnv },
|
|
992
|
+
private readonly agentEnv: { director: AgentEnv; worker: AgentEnv; crew?: Record<string, AgentEnv> },
|
|
887
993
|
/**
|
|
888
994
|
* Per-token rates per role, where the endpoint that will send the bill
|
|
889
995
|
* published them (see prices.ts). Absent for an Anthropic-native role —
|
|
@@ -908,7 +1014,11 @@ export class MissionRun {
|
|
|
908
1014
|
*/
|
|
909
1015
|
private readonly ledger?: {
|
|
910
1016
|
key: string;
|
|
911
|
-
roles: {
|
|
1017
|
+
roles: {
|
|
1018
|
+
director: boolean; worker: boolean;
|
|
1019
|
+
/** True when any crew preset runs on a gateway of its own — its tokens reach the same ledger. */
|
|
1020
|
+
crew?: boolean;
|
|
1021
|
+
};
|
|
912
1022
|
read: (key: string) => Promise<TokenUsage & { calls: number; costUsd?: number } | null>;
|
|
913
1023
|
},
|
|
914
1024
|
/**
|
|
@@ -917,7 +1027,19 @@ export class MissionRun {
|
|
|
917
1027
|
* before another token is spent. Absent means "unknown", which leaves the
|
|
918
1028
|
* run's basis alone.
|
|
919
1029
|
*/
|
|
920
|
-
private readonly roleBasis?: {
|
|
1030
|
+
private readonly roleBasis?: {
|
|
1031
|
+
director: CostBasis; worker: CostBasis;
|
|
1032
|
+
/**
|
|
1033
|
+
* A crew preset that runs on a provider of its own, by preset id. Its
|
|
1034
|
+
* review is not free just because the run's workers are: a paid
|
|
1035
|
+
* reviewer beside a free gateway worker had its dollars dropped, because
|
|
1036
|
+
* cost was attributed to the worker role and that role's figures are
|
|
1037
|
+
* discarded. What a provider will bill is decided by the provider, so
|
|
1038
|
+
* the preset's own basis is what counts its spend and what folds into
|
|
1039
|
+
* the run's.
|
|
1040
|
+
*/
|
|
1041
|
+
crew?: Record<string, { basis: CostBasis; native: boolean }>;
|
|
1042
|
+
},
|
|
921
1043
|
/**
|
|
922
1044
|
* What the host lends the run beyond the model: today, putting a dev
|
|
923
1045
|
* server the crew started behind Foreman's own address so the human can
|
|
@@ -1161,7 +1283,7 @@ export class MissionRun {
|
|
|
1161
1283
|
// Only a run with a gatewayed role has anything to poll for. Three
|
|
1162
1284
|
// seconds is chosen against what it is for — a person watching a turn
|
|
1163
1285
|
// that has been silent for minutes — not against how fast tokens move.
|
|
1164
|
-
if (this.ledger && (this.ledger.roles.director || this.ledger.roles.worker)) {
|
|
1286
|
+
if (this.ledger && (this.ledger.roles.director || this.ledger.roles.worker || this.ledger.roles.crew)) {
|
|
1165
1287
|
this.ledgerTimer = setInterval(() => void this.pollLedger(), 3000);
|
|
1166
1288
|
this.ledgerTimer.unref?.();
|
|
1167
1289
|
}
|
|
@@ -1204,10 +1326,12 @@ export class MissionRun {
|
|
|
1204
1326
|
'satisfy their milestone. Update the doc to match reality, then continue ' +
|
|
1205
1327
|
'the mission to DONE WHEN. ' +
|
|
1206
1328
|
this.gitLine() +
|
|
1329
|
+
this.crewLine() +
|
|
1207
1330
|
this.budgetNote() +
|
|
1208
1331
|
memorySection((await readMemory(this.meta.folder)).text, 'director')
|
|
1209
1332
|
: `MISSION: ${this.meta.mission}\n\n${this.budgetLine()} ` +
|
|
1210
1333
|
`Working directory: ${this.meta.folder}. ${this.gitLine()}Begin by writing .foreman/MISSION.md, then execute the plan.` +
|
|
1334
|
+
this.crewLine() +
|
|
1211
1335
|
memorySection((await readMemory(this.meta.folder)).text, 'director');
|
|
1212
1336
|
|
|
1213
1337
|
try {
|
|
@@ -1362,22 +1486,57 @@ export class MissionRun {
|
|
|
1362
1486
|
await this.ignoreLocalSettings();
|
|
1363
1487
|
if (this.meta.status === 'running') this.meta.status = 'interrupted';
|
|
1364
1488
|
|
|
1365
|
-
|
|
1366
|
-
|
|
1367
|
-
|
|
1368
|
-
|
|
1369
|
-
|
|
1370
|
-
|
|
1371
|
-
|
|
1372
|
-
|
|
1373
|
-
|
|
1374
|
-
|
|
1375
|
-
|
|
1376
|
-
|
|
1377
|
-
|
|
1378
|
-
|
|
1379
|
-
|
|
1380
|
-
|
|
1489
|
+
await this.settleFinalStatus();
|
|
1490
|
+
this.meta.endedAt = Date.now();
|
|
1491
|
+
this.saveMeta(this.meta);
|
|
1492
|
+
this.emit('run_finished', {
|
|
1493
|
+
status: this.meta.status,
|
|
1494
|
+
costUsd: this.meta.costUsd,
|
|
1495
|
+
directorSessionId: this.meta.directorSessionId,
|
|
1496
|
+
});
|
|
1497
|
+
}
|
|
1498
|
+
}
|
|
1499
|
+
|
|
1500
|
+
/**
|
|
1501
|
+
* The last word on whether this run counts as done — the two records that
|
|
1502
|
+
* outrank the director's own exit, applied in one place.
|
|
1503
|
+
*
|
|
1504
|
+
* A director's exit is not proof its mission succeeded. The charter makes
|
|
1505
|
+
* it write DONE WHEN criteria and tick each one the moment it is actually
|
|
1506
|
+
* verified, so criteria still unticked at exit are the director's own
|
|
1507
|
+
* record that the work is unfinished — and reporting that as 'done' is
|
|
1508
|
+
* the one lie a mission runner cannot afford. Downgrading to
|
|
1509
|
+
* 'interrupted' is also the useful answer: it is what makes the run
|
|
1510
|
+
* resumable rather than closed.
|
|
1511
|
+
* A mission whose own record says every criterion is verified is done,
|
|
1512
|
+
* even if the cap ended the turn it was writing its report in. Calling
|
|
1513
|
+
* that 'interrupted' told the fleet a finished mission had failed, and
|
|
1514
|
+
* invited a resume that spent more to rewrite a report already on disk.
|
|
1515
|
+
* The stop reason stays on the record, so nothing is hidden.
|
|
1516
|
+
*
|
|
1517
|
+
* Separated from `start`'s `finally` because it is the rule, not the
|
|
1518
|
+
* teardown: a test can drive it on a run whose status and crew are set by
|
|
1519
|
+
* hand, which is the only way the gate below is provable without an SDK.
|
|
1520
|
+
*/
|
|
1521
|
+
private async settleFinalStatus(): Promise<void> {
|
|
1522
|
+
// The reviewer gate, computed once for the whole method: whether every
|
|
1523
|
+
// reviewer the human marked required has passed the diff this run
|
|
1524
|
+
// actually ends with. Null means nothing is outstanding — either no
|
|
1525
|
+
// required reviewer, or all of them passed the current diff. It gates
|
|
1526
|
+
// both routes to 'done' below, because a run that reached its cap with
|
|
1527
|
+
// every box ticked is in exactly the position the gate exists for: the
|
|
1528
|
+
// director says it is finished and nobody else has looked.
|
|
1529
|
+
const gate = await this.reviewGate();
|
|
1530
|
+
if (this.meta.status === 'interrupted' && this.budgetStopped
|
|
1531
|
+
&& !this.wasInterrupted && !this.usageLimited) {
|
|
1532
|
+
const unmet = await this.unmetCriteria();
|
|
1533
|
+
if (doneAtCap({ budgetStopped: this.budgetStopped, wasInterrupted: this.wasInterrupted, usageLimited: this.usageLimited }, unmet)) {
|
|
1534
|
+
if (gate) {
|
|
1535
|
+
// Ticked every box and ran out of budget, but the reviewer the
|
|
1536
|
+
// human required never passed this diff: the run stays interrupted
|
|
1537
|
+
// and resumable, and the event says which reviewer and why.
|
|
1538
|
+
this.emit('mission_unreviewed', { blockers: gate.blockers, text: gate.text });
|
|
1539
|
+
} else {
|
|
1381
1540
|
this.meta.status = 'done';
|
|
1382
1541
|
this.emit('mission_done_at_cap', {
|
|
1383
1542
|
costUsd: this.meta.costUsd, budgetUsd: this.meta.budgetUsd, reason: this.meta.stopReason,
|
|
@@ -1386,25 +1545,24 @@ export class MissionRun {
|
|
|
1386
1545
|
});
|
|
1387
1546
|
}
|
|
1388
1547
|
}
|
|
1389
|
-
|
|
1390
|
-
|
|
1391
|
-
|
|
1392
|
-
|
|
1393
|
-
|
|
1394
|
-
|
|
1395
|
-
|
|
1396
|
-
|
|
1397
|
-
|
|
1398
|
-
|
|
1399
|
-
}
|
|
1548
|
+
}
|
|
1549
|
+
if (this.meta.status === 'done') {
|
|
1550
|
+
const unmet = await this.unmetCriteria();
|
|
1551
|
+
if (unmet?.length) {
|
|
1552
|
+
this.meta.status = 'interrupted';
|
|
1553
|
+
this.emit('mission_incomplete', {
|
|
1554
|
+
unmet,
|
|
1555
|
+
text: `The director ended with ${unmet.length} DONE WHEN criteri` +
|
|
1556
|
+
`${unmet.length === 1 ? 'on' : 'a'} still unticked, so this run is not done. ` +
|
|
1557
|
+
'Resume to continue it.',
|
|
1558
|
+
});
|
|
1559
|
+
} else if (gate) {
|
|
1560
|
+
// Every box ticked and the director says it is finished — and the
|
|
1561
|
+
// reviewer the human required has not passed the diff it ends with.
|
|
1562
|
+
// Foreman refuses the label rather than asking the director to.
|
|
1563
|
+
this.meta.status = 'interrupted';
|
|
1564
|
+
this.emit('mission_unreviewed', { blockers: gate.blockers, text: gate.text });
|
|
1400
1565
|
}
|
|
1401
|
-
this.meta.endedAt = Date.now();
|
|
1402
|
-
this.saveMeta(this.meta);
|
|
1403
|
-
this.emit('run_finished', {
|
|
1404
|
-
status: this.meta.status,
|
|
1405
|
-
costUsd: this.meta.costUsd,
|
|
1406
|
-
directorSessionId: this.meta.directorSessionId,
|
|
1407
|
-
});
|
|
1408
1566
|
}
|
|
1409
1567
|
}
|
|
1410
1568
|
|
|
@@ -1443,19 +1601,13 @@ export class MissionRun {
|
|
|
1443
1601
|
* whose director never wrote a doc has already failed more visibly.
|
|
1444
1602
|
*/
|
|
1445
1603
|
private async unmetCriteria(): Promise<string[] | null> {
|
|
1446
|
-
const doc = await
|
|
1447
|
-
|
|
1448
|
-
if (
|
|
1449
|
-
|
|
1450
|
-
const lines = doc.split('\n');
|
|
1451
|
-
const start = lines.findIndex((l) => /^#{1,6}\s*DONE\s*WHEN/i.test(l.trim()));
|
|
1452
|
-
if (start === -1) return null;
|
|
1604
|
+
const doc = await this.missionDoc();
|
|
1605
|
+
const section = doc === null ? null : doneWhenSection(doc);
|
|
1606
|
+
if (section === null) return null;
|
|
1453
1607
|
|
|
1454
1608
|
const unmet: string[] = [];
|
|
1455
1609
|
let sawAny = false;
|
|
1456
|
-
for (const line of
|
|
1457
|
-
// The section ends at the next heading; checkboxes below it are the plan.
|
|
1458
|
-
if (/^#{1,6}\s/.test(line)) break;
|
|
1610
|
+
for (const line of section) {
|
|
1459
1611
|
const box = line.match(/^\s*[-*]\s*\[( |x|X)\]\s*(.*)$/);
|
|
1460
1612
|
if (!box) continue;
|
|
1461
1613
|
sawAny = true;
|
|
@@ -1464,6 +1616,11 @@ export class MissionRun {
|
|
|
1464
1616
|
return sawAny ? unmet : null;
|
|
1465
1617
|
}
|
|
1466
1618
|
|
|
1619
|
+
/** The mission doc as text, or null when the director never wrote one. */
|
|
1620
|
+
private missionDoc(): Promise<string | null> {
|
|
1621
|
+
return readFile(path.join(this.meta.folder, '.foreman', 'MISSION.md'), 'utf8').catch(() => null);
|
|
1622
|
+
}
|
|
1623
|
+
|
|
1467
1624
|
/**
|
|
1468
1625
|
* A headless Playwright browser (its own profile — never the user's
|
|
1469
1626
|
* Chrome), granted when the mission enabled browser tools.
|
|
@@ -1483,7 +1640,14 @@ export class MissionRun {
|
|
|
1483
1640
|
};
|
|
1484
1641
|
}
|
|
1485
1642
|
|
|
1486
|
-
|
|
1643
|
+
/**
|
|
1644
|
+
* The permission callback for one agent. `override` tightens (or loosens)
|
|
1645
|
+
* the run's tool policy for that agent alone — see
|
|
1646
|
+
* {@link WorkerOverrides.toolPolicy}. Merged OVER the run's policy rather
|
|
1647
|
+
* than replacing it, so a reviewer still inherits everything the run
|
|
1648
|
+
* decided and only differs where the preset says it must.
|
|
1649
|
+
*/
|
|
1650
|
+
private policyFor(agent: string, override?: ToolPolicy) {
|
|
1487
1651
|
return makePolicy(agent, this.meta.folder, this.runAllowed, this.allowedRoots, {
|
|
1488
1652
|
onAutoAllow: (a, toolName, reason) => this.emit('auto_allowed', { agent: a, toolName, reason }),
|
|
1489
1653
|
onAutoDeny: (a, toolName, reason) => this.emit('auto_denied', { agent: a, toolName, reason }),
|
|
@@ -1514,7 +1678,10 @@ export class MissionRun {
|
|
|
1514
1678
|
if (existed) this.emit('permission_resolved', { id, behavior: 'aborted' });
|
|
1515
1679
|
return existed;
|
|
1516
1680
|
},
|
|
1517
|
-
}, {
|
|
1681
|
+
}, {
|
|
1682
|
+
toolPolicy: override ? { ...this.meta.toolPolicy, ...override } : this.meta.toolPolicy,
|
|
1683
|
+
autoAllowReadOnly: this.meta.autoAllowReadOnly,
|
|
1684
|
+
});
|
|
1518
1685
|
}
|
|
1519
1686
|
|
|
1520
1687
|
/**
|
|
@@ -1569,8 +1736,19 @@ export class MissionRun {
|
|
|
1569
1736
|
this.meta.costParts = { ...p };
|
|
1570
1737
|
}
|
|
1571
1738
|
|
|
1572
|
-
private addCost(usd: number | undefined, role: AgentRole = 'director'): void {
|
|
1739
|
+
private addCost(usd: number | undefined, role: AgentRole = 'director', native = false): void {
|
|
1573
1740
|
if (typeof usd !== 'number') return;
|
|
1741
|
+
// `native` is for an agent that runs somewhere else entirely — a crew
|
|
1742
|
+
// preset on its own priced provider — where the role's treatment is the
|
|
1743
|
+
// wrong answer and the SDK's dollars are a fact about this run.
|
|
1744
|
+
if (native) {
|
|
1745
|
+
this.costParts.native += usd;
|
|
1746
|
+
this.recomputeCost();
|
|
1747
|
+
this.saveMeta(this.meta);
|
|
1748
|
+
this.emitEconomics();
|
|
1749
|
+
this.enforceBudget();
|
|
1750
|
+
return;
|
|
1751
|
+
}
|
|
1574
1752
|
// Two cases where the SDK's dollar figure is not a fact about this run:
|
|
1575
1753
|
// a role Foreman prices itself, and a role behind a gateway at all. The
|
|
1576
1754
|
// second is the one that leaked — Anthropic's table applied to 5.8M
|
|
@@ -1665,7 +1843,7 @@ export class MissionRun {
|
|
|
1665
1843
|
};
|
|
1666
1844
|
}
|
|
1667
1845
|
|
|
1668
|
-
private addUsage(raw: unknown, role: AgentRole = 'director'): void {
|
|
1846
|
+
private addUsage(raw: unknown, role: AgentRole = 'director', priced = false): void {
|
|
1669
1847
|
// The real number for this turn just arrived; the estimate standing in
|
|
1670
1848
|
// for it is now redundant, and the ledger must be re-baselined so the
|
|
1671
1849
|
// same tokens are not offered again as growth.
|
|
@@ -1677,7 +1855,13 @@ export class MissionRun {
|
|
|
1677
1855
|
// Where the endpoint published rates, this is the run's real cost: its
|
|
1678
1856
|
// own tokens at its own prices, accumulated per role so a mixed run bills
|
|
1679
1857
|
// each half correctly instead of applying one table to both.
|
|
1680
|
-
|
|
1858
|
+
// `priced` means this agent runs on a provider of its own: either an
|
|
1859
|
+
// Anthropic-native one whose dollars addCost has already taken, or a
|
|
1860
|
+
// gateway that reports through the run's ledger. Either way the ROLE's
|
|
1861
|
+
// rate card is the wrong table — it belongs to a different endpoint — and
|
|
1862
|
+
// applying it would charge tokens that are already accounted for. The
|
|
1863
|
+
// tokens themselves still count.
|
|
1864
|
+
const price = priced ? undefined : this.prices[role];
|
|
1681
1865
|
if (price) { this.costParts.rated += priceUsage(price, delta); this.recomputeCost(); }
|
|
1682
1866
|
this.saveMeta(this.meta);
|
|
1683
1867
|
this.emitEconomics();
|
|
@@ -1745,6 +1929,31 @@ export class MissionRun {
|
|
|
1745
1929
|
}
|
|
1746
1930
|
|
|
1747
1931
|
/** The mission's branch, when it has one: stay on it, and leave merging and pushing alone. */
|
|
1932
|
+
/**
|
|
1933
|
+
* Who is on this run's crew, by id, and which of them must pass before the
|
|
1934
|
+
* mission can be recorded as done.
|
|
1935
|
+
*
|
|
1936
|
+
* Without this the charter asks for a review the director has no way to
|
|
1937
|
+
* name: the ids live in `meta.crew`, which no prompt showed, so the only
|
|
1938
|
+
* route to them was calling the tool with a wrong id and reading the error.
|
|
1939
|
+
* A gate the director cannot see is a gate it fails by accident.
|
|
1940
|
+
*/
|
|
1941
|
+
private crewLine(): string {
|
|
1942
|
+
const crew = this.meta.crew ?? [];
|
|
1943
|
+
if (!crew.length) return '';
|
|
1944
|
+
const rows = crew.map((p) => ` - ${p.id} — ${p.name}${p.model ? ` (${p.model})` : ''}: `
|
|
1945
|
+
+ (p.requiredForDone
|
|
1946
|
+
? 'REQUIRED. This run cannot be recorded as done until it returns VERDICT: PASS on the work as it finally stands.'
|
|
1947
|
+
: 'optional; ask for it when its subject is in play.')).join('\n');
|
|
1948
|
+
const required = crew.filter((p) => p.requiredForDone);
|
|
1949
|
+
return `\n\nYOUR CREW — call mcp__foreman__request_review with one of these ids:\n${rows}\n`
|
|
1950
|
+
+ (required.length
|
|
1951
|
+
? 'Request the required review AFTER your last change and before you tick the final box: a PASS is '
|
|
1952
|
+
+ 'pinned to the work as it was reviewed, so anything you change afterwards makes it stale and the '
|
|
1953
|
+
+ 'run reads as not done. If a review comes back FAIL, fix what it found and request it again.\n'
|
|
1954
|
+
: '');
|
|
1955
|
+
}
|
|
1956
|
+
|
|
1748
1957
|
private gitLine(): string {
|
|
1749
1958
|
const g = this.meta.git;
|
|
1750
1959
|
if (!g) return '';
|
|
@@ -1983,6 +2192,14 @@ export class MissionRun {
|
|
|
1983
2192
|
const nextId = `worker-${++this.workerSeq}`;
|
|
1984
2193
|
this.launchWorker(nextId, prompt, undefined, {
|
|
1985
2194
|
env: this.agentEnv.director, model: this.meta.directorModel || undefined, priceRole: 'director',
|
|
2195
|
+
// A reviewer retried elsewhere is still a reviewer. Changing which
|
|
2196
|
+
// provider serves it must not hand it Write, Edit and Bash: the
|
|
2197
|
+
// restriction belongs to the job, not to the endpoint.
|
|
2198
|
+
toolPolicy: this.workers.get(workerId)?.overrides?.toolPolicy,
|
|
2199
|
+
// Including the browser: a reviewer retried elsewhere must not gain
|
|
2200
|
+
// clicks and injected script by changing endpoint.
|
|
2201
|
+
noBrowser: this.workers.get(workerId)?.overrides?.noBrowser,
|
|
2202
|
+
crewPresetId: this.workers.get(workerId)?.crewPresetId,
|
|
1986
2203
|
reason: `retry of ${workerId} on the director's provider, at the human's request`,
|
|
1987
2204
|
});
|
|
1988
2205
|
return {
|
|
@@ -2025,6 +2242,10 @@ export class MissionRun {
|
|
|
2025
2242
|
w.isError = undefined;
|
|
2026
2243
|
w.reportShown = false;
|
|
2027
2244
|
w.toolCalls ??= 0;
|
|
2245
|
+
// Kept for the resume path: see WorkerRuntime.overrides. A resume passes
|
|
2246
|
+
// these back in, so `overrides ?? w.overrides` is what a follow-up uses.
|
|
2247
|
+
if (overrides) w.overrides = overrides;
|
|
2248
|
+
if (overrides?.crewPresetId) w.crewPresetId = overrides.crewPresetId;
|
|
2028
2249
|
w.recent ??= [];
|
|
2029
2250
|
w.done = new Promise<void>((resolve) => { w.settle = resolve; });
|
|
2030
2251
|
this.workers.set(workerId, w);
|
|
@@ -2079,8 +2300,8 @@ export class MissionRun {
|
|
|
2079
2300
|
// Built per worker: the report_progress handler closes over this id,
|
|
2080
2301
|
// which is how a report lands on the right record without the worker
|
|
2081
2302
|
// having to know its own name.
|
|
2082
|
-
mcpServers: { foreman: this.workerTools(workerId), ...this.browserServers() },
|
|
2083
|
-
canUseTool: this.policyFor(workerId),
|
|
2303
|
+
mcpServers: { foreman: this.workerTools(workerId), ...(overrides?.noBrowser ? {} : this.browserServers()) },
|
|
2304
|
+
canUseTool: this.policyFor(workerId, overrides?.toolPolicy),
|
|
2084
2305
|
},
|
|
2085
2306
|
});
|
|
2086
2307
|
w.q = q;
|
|
@@ -2138,8 +2359,9 @@ export class MissionRun {
|
|
|
2138
2359
|
// usage, so usage has to be current before it is emitted.
|
|
2139
2360
|
// A worker retried on the director's provider is priced with the
|
|
2140
2361
|
// director's rates: the tokens went through that gateway.
|
|
2141
|
-
this.addUsage(m.usage, overrides?.priceRole ?? 'worker'
|
|
2142
|
-
|
|
2362
|
+
this.addUsage(m.usage, overrides?.priceRole ?? 'worker',
|
|
2363
|
+
overrides?.nativeCost === true || overrides?.ownProvider === true);
|
|
2364
|
+
this.addCost(m.total_cost_usd as number | undefined, overrides?.priceRole ?? 'worker', overrides?.nativeCost === true);
|
|
2143
2365
|
w.costUsd += (m.total_cost_usd as number | undefined) ?? 0;
|
|
2144
2366
|
}
|
|
2145
2367
|
this.emit('message', { agent: workerId, msg });
|
|
@@ -2304,6 +2526,210 @@ export class MissionRun {
|
|
|
2304
2526
|
`to watch it, and wait_for_worker when you need its result.`;
|
|
2305
2527
|
}
|
|
2306
2528
|
|
|
2529
|
+
/**
|
|
2530
|
+
* request_review — hand the run's diff to a crew reviewer and wait for its
|
|
2531
|
+
* verdict.
|
|
2532
|
+
*
|
|
2533
|
+
* Unlike spawn_worker this AWAITS: the director asked a question and the
|
|
2534
|
+
* answer is the whole point of the call. The reviewer is an ordinary worker
|
|
2535
|
+
* in every other respect — same id sequence, same budget, same transcript —
|
|
2536
|
+
* because a review that did not count against the run's spend would be a
|
|
2537
|
+
* cost nobody could see, and a reviewer with a second kind of id would make
|
|
2538
|
+
* every other surface learn a shape it does not need.
|
|
2539
|
+
*
|
|
2540
|
+
* What it is NOT is a way to ask nicely. The verdict is recorded on the run
|
|
2541
|
+
* whatever the director does with it, and the gate in the end-of-run
|
|
2542
|
+
* `finally` reads that record, not this conversation.
|
|
2543
|
+
*/
|
|
2544
|
+
/**
|
|
2545
|
+
* The launch overrides for a worker that is a crew preset, rebuilt from the
|
|
2546
|
+
* run's frozen crew. Used when the record survived a resume but the live
|
|
2547
|
+
* overrides did not: they hold a credential and are deliberately not
|
|
2548
|
+
* persisted, while the restriction they carry must not lapse.
|
|
2549
|
+
*/
|
|
2550
|
+
private overridesForPreset(presetId?: string): WorkerOverrides | undefined {
|
|
2551
|
+
if (!presetId) return undefined;
|
|
2552
|
+
const preset = (this.meta.crew ?? []).find((p) => p.id === presetId);
|
|
2553
|
+
if (!preset) return undefined;
|
|
2554
|
+
return {
|
|
2555
|
+
model: preset.model || this.meta.workerModel,
|
|
2556
|
+
env: this.agentEnv.crew?.[preset.id],
|
|
2557
|
+
toolPolicy: preset.toolPolicy === 'default' ? undefined : REVIEWER_TOOL_POLICY,
|
|
2558
|
+
nativeCost: this.roleBasis?.crew?.[preset.id]?.native === true
|
|
2559
|
+
&& this.roleBasis.crew[preset.id].basis === 'priced'
|
|
2560
|
+
&& Boolean(this.agentEnv.crew?.[preset.id]),
|
|
2561
|
+
noBrowser: preset.toolPolicy !== 'default',
|
|
2562
|
+
ownProvider: Boolean(this.agentEnv.crew?.[preset.id]),
|
|
2563
|
+
crewPresetId: preset.id,
|
|
2564
|
+
reason: `follow-up to ${preset.name} (${preset.id})`,
|
|
2565
|
+
};
|
|
2566
|
+
}
|
|
2567
|
+
|
|
2568
|
+
private async requestReviewTool({ presetId, notes }: { presetId: string; notes?: string }): Promise<string> {
|
|
2569
|
+
const crew = this.meta.crew ?? [];
|
|
2570
|
+
if (!crew.length) {
|
|
2571
|
+
return 'This run has no crew: nobody was added as a reviewer when it was dispatched, so ' +
|
|
2572
|
+
'there is nothing to request a review from. Verify the work yourself and say so in your report.';
|
|
2573
|
+
}
|
|
2574
|
+
const preset = crew.find((p) => p.id === presetId);
|
|
2575
|
+
if (!preset) {
|
|
2576
|
+
return `No crew preset "${presetId}" on this run. Available: ${crew.map((p) => p.id).join(', ')}.`;
|
|
2577
|
+
}
|
|
2578
|
+
const stop = this.overBudget();
|
|
2579
|
+
if (stop) return stop;
|
|
2580
|
+
|
|
2581
|
+
// The diff as Foreman sees it — the same deck the gate will hash — so a
|
|
2582
|
+
// PASS is about the change Foreman will check, not about whatever the
|
|
2583
|
+
// reviewer happened to look at.
|
|
2584
|
+
// What the reviewer READS is the deck, which is a capped view. What the
|
|
2585
|
+
// PASS is PINNED TO is the working tree itself — see changeFingerprint.
|
|
2586
|
+
// A review that cannot be pinned is not worth having: it would clear the
|
|
2587
|
+
// gate for a diff nobody can prove was the one read.
|
|
2588
|
+
const fingerprint = await changeFingerprint(this.meta.folder);
|
|
2589
|
+
if (fingerprint === null) {
|
|
2590
|
+
return 'Foreman cannot read what this run has changed (the folder is not a git repository, or git ' +
|
|
2591
|
+
'would not answer), so a review cannot be pinned to it and would not clear the gate. Verify the ' +
|
|
2592
|
+
'work yourself, say so plainly in your report, and record in MISSION.md that no review was possible.';
|
|
2593
|
+
}
|
|
2594
|
+
const deck = await deckFor(this.meta.folder, this.meta.id);
|
|
2595
|
+
if (deck.baseline?.kind === 'none') {
|
|
2596
|
+
return 'Foreman has no baseline for this run, so it cannot show a reviewer what changed — the deck ' +
|
|
2597
|
+
'is empty whatever the crew did. A PASS on nothing would clear the gate on nothing, so no review ' +
|
|
2598
|
+
'is requested. Say this in your report.';
|
|
2599
|
+
}
|
|
2600
|
+
const { diff, truncated } = renderDeckDiff(deck);
|
|
2601
|
+
const doc = await this.missionDoc();
|
|
2602
|
+
const doneWhen = (doc === null ? null : doneWhenSection(doc))?.join('\n') ?? '';
|
|
2603
|
+
let prompt = reviewBriefFor(preset, { mission: this.meta.mission, doneWhen, diff, truncated });
|
|
2604
|
+
if (notes?.trim()) prompt += `\n\n## The director asks you to pay particular attention to\n\n${notes.trim()}`;
|
|
2605
|
+
|
|
2606
|
+
// A reviewer on a priced provider spends real money even when the rest of
|
|
2607
|
+
// the run does not, and a dollar cap that is not "live" is not enforced at
|
|
2608
|
+
// all — enforceBudget and capReached both stand down on an unpriced run.
|
|
2609
|
+
// So the basis moves BEFORE the reviewer is launched, and says so, exactly
|
|
2610
|
+
// as the fallback path does when a human retries on the director's.
|
|
2611
|
+
const presetCost = this.agentEnv.crew?.[preset.id] ? this.roleBasis?.crew?.[preset.id] : undefined;
|
|
2612
|
+
if (presetCost) {
|
|
2613
|
+
const nb = combineBasis(costBasisOf(this.meta), presetCost.basis);
|
|
2614
|
+
if (nb !== costBasisOf(this.meta)) {
|
|
2615
|
+
this.meta.costBasis = nb;
|
|
2616
|
+
this.meta.metered = nb === 'priced';
|
|
2617
|
+
this.emit('settings_changed', {
|
|
2618
|
+
changes: [`${preset.name} reviews on its own provider — this run is now ${nb}`
|
|
2619
|
+
+ (nb === 'priced' ? ' and the dollar cap is live' : '')],
|
|
2620
|
+
browserTools: Boolean(this.meta.browserTools), budgetUsd: this.meta.budgetUsd,
|
|
2621
|
+
});
|
|
2622
|
+
this.saveMeta(this.meta);
|
|
2623
|
+
this.emitEconomics();
|
|
2624
|
+
}
|
|
2625
|
+
}
|
|
2626
|
+
|
|
2627
|
+
const id = `worker-${++this.workerSeq}`;
|
|
2628
|
+
const w = this.launchWorker(id, prompt, undefined, {
|
|
2629
|
+
model: preset.model || this.meta.workerModel,
|
|
2630
|
+
// The preset's own provider, when dispatch managed to resolve one for it;
|
|
2631
|
+
// undefined leaves runWorker on the worker env. A reviewer is still worth
|
|
2632
|
+
// running on a fallback provider — refusing to review because a provider
|
|
2633
|
+
// was deleted after the crew was chosen would strand the run at the one
|
|
2634
|
+
// step that lets it be called done.
|
|
2635
|
+
//
|
|
2636
|
+
// Its tokens are still priced at the WORKER role's rates, deliberately.
|
|
2637
|
+
// Per-preset pricing would mean a fourth rate card and a fourth gateway
|
|
2638
|
+
// bucket for a handful of turns; a mixed run already resolves to the
|
|
2639
|
+
// dearer basis rather than the cheaper one (see combineBasis in types.ts),
|
|
2640
|
+
// so the error this leaves behind overstates spend, which is the side a
|
|
2641
|
+
// cost cap should be wrong on.
|
|
2642
|
+
env: this.agentEnv.crew?.[preset.id],
|
|
2643
|
+
// A preset with an env of its own is served by that provider, so its
|
|
2644
|
+
// basis — not the worker role's — decides whether its dollars are real.
|
|
2645
|
+
// Only an Anthropic-native endpoint gives the SDK a dollar figure that
|
|
2646
|
+
// IS the bill. A priced gateway publishes rates and reports through the
|
|
2647
|
+
// shared ledger, so counting the SDK's number there as well would charge
|
|
2648
|
+
// the review twice — the very thing the last fix was about.
|
|
2649
|
+
nativeCost: presetCost?.native === true && presetCost.basis === 'priced',
|
|
2650
|
+
// Its tokens go through its own endpoint, so the worker's rate card
|
|
2651
|
+
// does not describe them whatever the wire is.
|
|
2652
|
+
ownProvider: Boolean(this.agentEnv.crew?.[preset.id]),
|
|
2653
|
+
// A reviewer that must not write must not drive the browser either:
|
|
2654
|
+
// clicks, typing and injected script are automatically permitted once
|
|
2655
|
+
// browser tools are on, so a read-only verdict could change the app it
|
|
2656
|
+
// is judging. The restriction is about the job, not about the file system.
|
|
2657
|
+
noBrowser: preset.toolPolicy !== 'default',
|
|
2658
|
+
// 'default' is the preset saying this role needs to run things; anything
|
|
2659
|
+
// else gets the flat deny, which is what makes "read-only" provable.
|
|
2660
|
+
toolPolicy: preset.toolPolicy === 'default' ? undefined : REVIEWER_TOOL_POLICY,
|
|
2661
|
+
crewPresetId: preset.id,
|
|
2662
|
+
reason: `review by ${preset.name} (${preset.id})`,
|
|
2663
|
+
});
|
|
2664
|
+
const outcome = await w.promise;
|
|
2665
|
+
|
|
2666
|
+
const parsed = parseVerdict(outcome?.report ?? '');
|
|
2667
|
+
const verdict: ReviewVerdict = {
|
|
2668
|
+
presetId: preset.id,
|
|
2669
|
+
name: preset.name,
|
|
2670
|
+
// No verdict line is not a pass. The reviewer may have run out of turns
|
|
2671
|
+
// or answered in prose; either way nobody has said this diff is good,
|
|
2672
|
+
// and the gate must not treat silence as consent.
|
|
2673
|
+
pass: parsed?.pass ?? false,
|
|
2674
|
+
findings: parsed?.findings
|
|
2675
|
+
?? `${preset.name} never stated a verdict. Its report ended without the required ` +
|
|
2676
|
+
`\`VERDICT: PASS\`/\`VERDICT: FAIL\` line, so this counts as a FAIL.` +
|
|
2677
|
+
(outcome?.report ? `\n\nWhat it did report:\n${outcome.report}` : ''),
|
|
2678
|
+
diffHash: fingerprint,
|
|
2679
|
+
workerId: id,
|
|
2680
|
+
costUsd: this.workers.get(id)?.costUsd,
|
|
2681
|
+
at: Date.now(),
|
|
2682
|
+
};
|
|
2683
|
+
this.meta.reviews = [...(this.meta.reviews ?? []), verdict];
|
|
2684
|
+
this.saveMeta(this.meta);
|
|
2685
|
+
this.emit('review_verdict', {
|
|
2686
|
+
presetId: verdict.presetId, name: verdict.name, model: preset.model || this.meta.workerModel,
|
|
2687
|
+
pass: verdict.pass, findings: verdict.findings, diffHash: verdict.diffHash,
|
|
2688
|
+
workerId: verdict.workerId, costUsd: verdict.costUsd,
|
|
2689
|
+
});
|
|
2690
|
+
|
|
2691
|
+
const head = parsed === null
|
|
2692
|
+
? `[${preset.name} (${id}) returned NO VERDICT — recorded as a FAIL]`
|
|
2693
|
+
: `[${preset.name} (${id}) — VERDICT: ${verdict.pass ? 'PASS' : 'FAIL'}]`;
|
|
2694
|
+
const tail = verdict.pass
|
|
2695
|
+
? 'This PASS is pinned to the diff as it stands right now. If you change anything after ' +
|
|
2696
|
+
'this, the PASS goes stale and the run will not be recorded done — so review last.'
|
|
2697
|
+
: 'This review must pass before the run can be recorded done. Fix what it names, then call ' +
|
|
2698
|
+
'request_review again.';
|
|
2699
|
+
return `${head}\n${verdict.findings || '(no findings given)'}\n\n${tail}${this.costFooter()}`;
|
|
2700
|
+
}
|
|
2701
|
+
|
|
2702
|
+
/**
|
|
2703
|
+
* THE GATE, as the end of the run sees it: null when nothing required is
|
|
2704
|
+
* outstanding, otherwise the blockers and the sentence for the event.
|
|
2705
|
+
*
|
|
2706
|
+
* A deck that cannot be computed leaves the gate UNSATISFIED rather than
|
|
2707
|
+
* open. "We could not work out what this run changed" is not evidence that
|
|
2708
|
+
* what it changed was reviewed, and the failure mode of the other reading —
|
|
2709
|
+
* a diff error quietly certifying a run as done — is the one this whole
|
|
2710
|
+
* mechanism exists to prevent.
|
|
2711
|
+
*/
|
|
2712
|
+
private async reviewGate(): Promise<{ blockers: ReviewBlocker[]; text: string } | null> {
|
|
2713
|
+
const required = (this.meta.crew ?? []).filter((p) => p.requiredForDone);
|
|
2714
|
+
if (!required.length) return null;
|
|
2715
|
+
// deckFor never throws: a missing baseline comes back as a deck with
|
|
2716
|
+
// `baseline.kind === 'none'` and no files, which would hash like a clean
|
|
2717
|
+
// tree and let any PASS stand. So the gate asks git directly, and treats
|
|
2718
|
+
// "cannot tell" as "not reviewed" rather than as "unchanged".
|
|
2719
|
+
const fingerprint = await changeFingerprint(this.meta.folder);
|
|
2720
|
+
if (fingerprint === null) {
|
|
2721
|
+
return {
|
|
2722
|
+
blockers: required.map((p) => ({ presetId: p.id, name: p.name, reason: 'missing' as const })),
|
|
2723
|
+
text: `This run's diff could not be computed, so no review can be matched to it. ` +
|
|
2724
|
+
`${required.map((p) => p.name).join(', ')} ` +
|
|
2725
|
+
`${required.length === 1 ? 'is' : 'are'} required to pass before this run is done, and ` +
|
|
2726
|
+
'nothing here shows that happened. This run is not done. Resume to continue it.',
|
|
2727
|
+
};
|
|
2728
|
+
}
|
|
2729
|
+
const blockers = reviewBlockers(this.meta.crew, this.meta.reviews, fingerprint);
|
|
2730
|
+
return blockers.length ? { blockers, text: unreviewedText(blockers) } : null;
|
|
2731
|
+
}
|
|
2732
|
+
|
|
2307
2733
|
private async messageWorkerTool({ worker_id, message }: { worker_id: string; message: string }): Promise<string> {
|
|
2308
2734
|
const stop = this.overBudget();
|
|
2309
2735
|
if (stop) return stop;
|
|
@@ -2314,7 +2740,12 @@ export class MissionRun {
|
|
|
2314
2740
|
if (w.status === 'running') {
|
|
2315
2741
|
return `${worker_id} is still running — wait_for_worker it first, then send the follow-up.`;
|
|
2316
2742
|
}
|
|
2317
|
-
|
|
2743
|
+
// A reviewer's read-only policy, model and provider live in the overrides
|
|
2744
|
+
// its first launch was given. Resuming without them would hand the
|
|
2745
|
+
// director a worker that may now write, on whatever model the run's
|
|
2746
|
+
// workers use — the restriction would quietly expire at the first
|
|
2747
|
+
// follow-up question.
|
|
2748
|
+
const run = this.launchWorker(worker_id, message, w.sessionId, w.overrides ?? this.overridesForPreset(w.crewPresetId));
|
|
2318
2749
|
await run.promise;
|
|
2319
2750
|
return `${this.reportLine(run)}${this.costFooter()}`;
|
|
2320
2751
|
}
|
|
@@ -2386,6 +2817,20 @@ export class MissionRun {
|
|
|
2386
2817
|
async (args) => text(this.spawnWorkerTool(args)),
|
|
2387
2818
|
);
|
|
2388
2819
|
|
|
2820
|
+
const requestReview = tool(
|
|
2821
|
+
'request_review',
|
|
2822
|
+
'Hand this run\'s diff to one of the crew\'s reviewers and get its verdict. Unlike ' +
|
|
2823
|
+
'spawn_worker this WAITS and returns the findings. The reviewer reads only — it cannot ' +
|
|
2824
|
+
'change your code — and its verdict is recorded on the run: a reviewer marked required ' +
|
|
2825
|
+
'must return PASS on the FINAL diff before Foreman will record the run as done, so call ' +
|
|
2826
|
+
'this after your last change, not before.',
|
|
2827
|
+
{
|
|
2828
|
+
presetId: z.string().describe('The crew preset id to review, e.g. "reviewer"'),
|
|
2829
|
+
notes: z.string().optional().describe('Anything specific you want it to look at, beyond its standing brief'),
|
|
2830
|
+
},
|
|
2831
|
+
async (args) => text(await this.requestReviewTool(args)),
|
|
2832
|
+
);
|
|
2833
|
+
|
|
2389
2834
|
const checkWorkers = tool(
|
|
2390
2835
|
'check_workers',
|
|
2391
2836
|
'Show every worker (or one): status, age, seconds since its last activity, tool calls ' +
|
|
@@ -2495,7 +2940,7 @@ export class MissionRun {
|
|
|
2495
2940
|
);
|
|
2496
2941
|
return createSdkMcpServer({
|
|
2497
2942
|
name: 'foreman',
|
|
2498
|
-
tools: [spawnWorker, checkWorkers, waitForWorker, messageWorker, askHuman, exposeService, remember],
|
|
2943
|
+
tools: [spawnWorker, requestReview, checkWorkers, waitForWorker, messageWorker, askHuman, exposeService, remember],
|
|
2499
2944
|
});
|
|
2500
2945
|
}
|
|
2501
2946
|
}
|