@amenophis1er/foreman 0.1.16 → 0.1.18

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -3,8 +3,8 @@
3
3
  *
4
4
  * Responsibilities:
5
5
  * - Spawn the director session with its charter and in-process MCP tools
6
- * (spawn_worker / check_workers / wait_for_worker / message_worker /
7
- * ask_human).
6
+ * (spawn_worker / request_review / check_workers / wait_for_worker /
7
+ * message_worker / ask_human).
8
8
  * - Run workers as separate, resumable SDK sessions, concurrently with the
9
9
  * director's own turn: spawn_worker returns as soon as the worker exists,
10
10
  * and the director reads progress and results back through check_workers
@@ -95,9 +95,14 @@ import type { AgentEnv } from './provider.js';
95
95
  import { generateRunTitle } from './title.js';
96
96
  import { combineBasis, costBasisOf, isPriced, type CostBasis } from './types.js';
97
97
  import { priceUsage, type ModelPrice } from './prices.js';
98
- import { captureBaseline } from './deck.js';
98
+ import { captureBaseline, deckFor, type Deck } from './deck.js';
99
+ import { changeFingerprint } from './gitwork.js';
99
100
  import { memorySection, readMemory, writeMemory } from './memory.js';
100
- import type { RunMeta, TokenUsage, WorkerMeta, WorkerProgress } from './types.js';
101
+ import {
102
+ REVIEWER_TOOL_POLICY, diffHash, parseVerdict, reviewBlockers, reviewBriefFor, unreviewedText,
103
+ type ReviewBlocker, type ReviewVerdict,
104
+ } from './crew.js';
105
+ import type { RunMeta, ToolPolicy, TokenUsage, WorkerMeta, WorkerProgress } from './types.js';
101
106
 
102
107
  /** A run's usage before its first `result` message. */
103
108
  function emptyUsage(): TokenUsage {
@@ -291,6 +296,64 @@ export function doneAtCap(
291
296
  return Array.isArray(unmet) && unmet.length === 0;
292
297
  }
293
298
 
299
+ /**
300
+ * The lines under the mission doc's DONE WHEN heading, to the next heading —
301
+ * or null when the doc has no such section.
302
+ *
303
+ * One parse, two readers: the end-of-run check that counts unticked boxes, and
304
+ * the reviewer's brief, which quotes the criteria verbatim so the reviewer
305
+ * judges the diff against the same contract Foreman judges the run against.
306
+ * Two parsers would eventually disagree about what "the criteria" are, and the
307
+ * one place that must not happen is the gate.
308
+ */
309
+ export function doneWhenSection(doc: string): string[] | null {
310
+ const lines = doc.split('\n');
311
+ const start = lines.findIndex((l) => /^#{1,6}\s*DONE\s*WHEN/i.test(l.trim()));
312
+ if (start === -1) return null;
313
+ const out: string[] = [];
314
+ for (const line of lines.slice(start + 1)) {
315
+ // The section ends at the next heading; checkboxes below it are the plan.
316
+ if (/^#{1,6}\s/.test(line)) break;
317
+ out.push(line);
318
+ }
319
+ return out;
320
+ }
321
+
322
+ /**
323
+ * How much of the run's diff fits in a reviewer's brief. The deck is already
324
+ * capped per file and per file count; this is the cap on the whole thing, so a
325
+ * mission that touched two hundred files does not hand its reviewer a prompt
326
+ * nothing will read to the end of. When it bites, the brief says so and the
327
+ * reviewer is told to open the files itself — it has Read, Glob and Grep.
328
+ */
329
+ export const REVIEW_DIFF_MAX_CHARS = 120_000;
330
+
331
+ /**
332
+ * A deck rendered as one diff text for the reviewer, with a header line per
333
+ * file so a truncated body is still attributable. `truncated` is true when
334
+ * this cut anything OR when the deck had already cut a file's own diff —
335
+ * either way the reviewer is looking at less than the whole change and must
336
+ * be told, because a reviewer that believes it saw everything passes on what
337
+ * it did not see.
338
+ */
339
+ export function renderDeckDiff(deck: Deck, max = REVIEW_DIFF_MAX_CHARS): { diff: string; truncated: boolean } {
340
+ const files = deck.files ?? [];
341
+ let truncated = files.length < (deck.totals?.files ?? files.length);
342
+ const parts: string[] = [];
343
+ let used = 0;
344
+ for (const f of files) {
345
+ if (f.truncated) truncated = true;
346
+ const head = `--- ${f.path} (${f.status} +${f.additions} -${f.deletions})`;
347
+ const body = f.binary ? '(binary file)' : (f.diff ?? '(no diff available)');
348
+ const block = `${head}\n${body}`;
349
+ if (used + block.length > max) { truncated = true; break; }
350
+ parts.push(block);
351
+ used += block.length + 1;
352
+ }
353
+ const diff = parts.join('\n') || (files.length ? '' : '(this run changed no files)');
354
+ return { diff, truncated };
355
+ }
356
+
294
357
  export function stopReasonOf(cap: string): 'budget' | 'turns' | 'time' | 'tokens' {
295
358
  if (cap.startsWith('TURN')) return 'turns';
296
359
  if (cap.startsWith('TIME')) return 'time';
@@ -631,9 +694,32 @@ type AgentRole = 'director' | 'worker';
631
694
  interface WorkerOverrides {
632
695
  /** Set on the one continuation a worker gets after the turn cap. */
633
696
  continued?: boolean;
697
+ /** The crew preset this worker is, recorded on its meta so a resume can rebuild this. */
698
+ crewPresetId?: string;
699
+ /** Withhold the browser from this worker, whatever the run allows. */
700
+ noBrowser?: boolean;
701
+ /** This agent runs on a provider of its own, so no role's rate card applies to it. */
702
+ ownProvider?: boolean;
703
+ /**
704
+ * Count this agent's dollars as real spend whatever the role it is priced
705
+ * under would normally do. Set for a crew preset on its own priced
706
+ * provider: its bill arrives regardless of how the run's workers are
707
+ * served, and a cap that cannot see it is not a cap.
708
+ */
709
+ nativeCost?: boolean;
634
710
  env?: AgentEnv;
635
711
  model?: string;
636
712
  priceRole?: AgentRole;
713
+ /**
714
+ * Tool rules for THIS worker only, merged over the run's own policy.
715
+ *
716
+ * A reviewer is a worker that must not be able to change what it is
717
+ * judging, and the run's policy cannot say that — it applies to the whole
718
+ * crew, and tightening it for everyone would stop the workers doing the
719
+ * work. So the restriction travels with the one agent it is about; every
720
+ * other worker and the director are built from the run's policy unchanged.
721
+ */
722
+ toolPolicy?: ToolPolicy;
637
723
  /** For the transcript and the report: why this worker exists. */
638
724
  reason?: string;
639
725
  }
@@ -738,25 +824,31 @@ directing worker agents. Non-negotiable rules, in priority order:
738
824
  after that the task comes back to you half-done. So give a worker a task it
739
825
  can finish in that many steps — split the big ones — rather than one brief
740
826
  that has to be rescued twice.
741
- 4. REPORT WHAT YOU SEE. Judge the work as a competent professional would, not
827
+ 4. GET THE REVIEW LAST. If this run has a crew reviewer marked required, call
828
+ mcp__foreman__request_review with its preset id AFTER your final change and
829
+ before you tick the last box. It reads the diff and answers PASS or FAIL;
830
+ Foreman will not record the run as done without that PASS, and a change made
831
+ after a PASS makes it stale — the reviewer would have passed code that no
832
+ longer exists — so review last, then stop.
833
+ 5. REPORT WHAT YOU SEE. Judge the work as a competent professional would, not
742
834
  only against the letter of the acceptance criteria. If you observe a defect
743
835
  the criteria did not name — tap targets too small to use, unreadable
744
836
  contrast, a broken layout, a hazard, an obviously wrong result — fix it
745
837
  when it is clearly in scope, and otherwise say so plainly in your final
746
838
  summary and in MISSION.md. Staying silent about a problem you could see is
747
839
  a failed mission even when every listed box is ticked.
748
- 5. DECIDE AND RECORD, DON'T ASK. You are running unattended more often than
840
+ 6. DECIDE AND RECORD, DON'T ASK. You are running unattended more often than
749
841
  not. Make the reasonable call, write it and the reasoning into MISSION.md,
750
842
  and continue. Reserve mcp__foreman__ask_human for decisions that are
751
843
  irreversible or that spend money the mission was not given — those you ask
752
844
  and wait for. A question left unanswered for ${DEFAULT_ASK_TIMEOUT_MS / 60_000}
753
845
  minutes is auto-answered "decide yourself"; treat that answer as the
754
846
  human's, record what you decided, and do not ask it again.
755
- 6. NEVER modify Foreman itself, its server, or any oversight tooling. Tooling
847
+ 7. NEVER modify Foreman itself, its server, or any oversight tooling. Tooling
756
848
  failure is an escalation, never a self-repair.
757
- 7. When DONE WHEN is verified, update MISSION.md (all boxes ticked, final log
849
+ 8. When DONE WHEN is verified, update MISSION.md (all boxes ticked, final log
758
850
  entry) and end with a short summary of what was built and how you verified it.
759
- 8. LEAVE NOTES FOR THE NEXT CREW. Before you finish, call mcp__foreman__remember
851
+ 9. LEAVE NOTES FOR THE NEXT CREW. Before you finish, call mcp__foreman__remember
760
852
  with the whole project memory as it should read now: how to run and test
761
853
  the project, ports and paths that matter, conventions and the reasons
762
854
  behind them, traps you fell into. Facts, one line each, a page at most.
@@ -805,6 +897,13 @@ const WORKER_CONTINUE_PROMPT =
805
897
  */
806
898
  interface WorkerRuntime extends WorkerMeta {
807
899
  q?: Query;
900
+ /**
901
+ * The overrides this worker was launched with, kept so a follow-up message
902
+ * resumes the same agent rather than an ordinary worker wearing its id — a
903
+ * reviewer's read-only policy, model and provider must outlive its first
904
+ * turn.
905
+ */
906
+ overrides?: WorkerOverrides;
808
907
  promise?: Promise<WorkerOutcome>;
809
908
  done?: Promise<void>;
810
909
  settle?: () => void;
@@ -812,7 +911,7 @@ interface WorkerRuntime extends WorkerMeta {
812
911
  reportShown?: boolean;
813
912
  }
814
913
 
815
- const RUNTIME_ONLY: ReadonlyArray<keyof WorkerRuntime> = ['q', 'promise', 'done', 'settle', 'reportShown'];
914
+ const RUNTIME_ONLY: ReadonlyArray<keyof WorkerRuntime> = ['q', 'promise', 'done', 'settle', 'reportShown', 'overrides'];
816
915
 
817
916
  export class MissionRun {
818
917
  readonly meta: RunMeta;
@@ -882,8 +981,15 @@ export class MissionRun {
882
981
  * provider model, and it is decided by which model each role was given.
883
982
  * Passed in rather than derived here so exactly one module decides what an
884
983
  * agent can authenticate as; see provider.ts.
984
+ *
985
+ * `crew` is the same thing for a crew preset that named a provider of its
986
+ * own, keyed by preset id, and it is deliberately partial: a preset with no
987
+ * `providerId`, or one whose provider no longer resolves, simply has no
988
+ * entry and runs on the worker env. Optional on the argument rather than a
989
+ * parameter of its own so every existing call site — and every test that
990
+ * constructs a run with two roles — keeps working untouched.
885
991
  */
886
- private readonly agentEnv: { director: AgentEnv; worker: AgentEnv },
992
+ private readonly agentEnv: { director: AgentEnv; worker: AgentEnv; crew?: Record<string, AgentEnv> },
887
993
  /**
888
994
  * Per-token rates per role, where the endpoint that will send the bill
889
995
  * published them (see prices.ts). Absent for an Anthropic-native role —
@@ -908,7 +1014,11 @@ export class MissionRun {
908
1014
  */
909
1015
  private readonly ledger?: {
910
1016
  key: string;
911
- roles: { director: boolean; worker: boolean };
1017
+ roles: {
1018
+ director: boolean; worker: boolean;
1019
+ /** True when any crew preset runs on a gateway of its own — its tokens reach the same ledger. */
1020
+ crew?: boolean;
1021
+ };
912
1022
  read: (key: string) => Promise<TokenUsage & { calls: number; costUsd?: number } | null>;
913
1023
  },
914
1024
  /**
@@ -917,7 +1027,19 @@ export class MissionRun {
917
1027
  * before another token is spent. Absent means "unknown", which leaves the
918
1028
  * run's basis alone.
919
1029
  */
920
- private readonly roleBasis?: { director: CostBasis; worker: CostBasis },
1030
+ private readonly roleBasis?: {
1031
+ director: CostBasis; worker: CostBasis;
1032
+ /**
1033
+ * A crew preset that runs on a provider of its own, by preset id. Its
1034
+ * review is not free just because the run's workers are: a paid
1035
+ * reviewer beside a free gateway worker had its dollars dropped, because
1036
+ * cost was attributed to the worker role and that role's figures are
1037
+ * discarded. What a provider will bill is decided by the provider, so
1038
+ * the preset's own basis is what counts its spend and what folds into
1039
+ * the run's.
1040
+ */
1041
+ crew?: Record<string, { basis: CostBasis; native: boolean }>;
1042
+ },
921
1043
  /**
922
1044
  * What the host lends the run beyond the model: today, putting a dev
923
1045
  * server the crew started behind Foreman's own address so the human can
@@ -1161,7 +1283,7 @@ export class MissionRun {
1161
1283
  // Only a run with a gatewayed role has anything to poll for. Three
1162
1284
  // seconds is chosen against what it is for — a person watching a turn
1163
1285
  // that has been silent for minutes — not against how fast tokens move.
1164
- if (this.ledger && (this.ledger.roles.director || this.ledger.roles.worker)) {
1286
+ if (this.ledger && (this.ledger.roles.director || this.ledger.roles.worker || this.ledger.roles.crew)) {
1165
1287
  this.ledgerTimer = setInterval(() => void this.pollLedger(), 3000);
1166
1288
  this.ledgerTimer.unref?.();
1167
1289
  }
@@ -1204,10 +1326,12 @@ export class MissionRun {
1204
1326
  'satisfy their milestone. Update the doc to match reality, then continue ' +
1205
1327
  'the mission to DONE WHEN. ' +
1206
1328
  this.gitLine() +
1329
+ this.crewLine() +
1207
1330
  this.budgetNote() +
1208
1331
  memorySection((await readMemory(this.meta.folder)).text, 'director')
1209
1332
  : `MISSION: ${this.meta.mission}\n\n${this.budgetLine()} ` +
1210
1333
  `Working directory: ${this.meta.folder}. ${this.gitLine()}Begin by writing .foreman/MISSION.md, then execute the plan.` +
1334
+ this.crewLine() +
1211
1335
  memorySection((await readMemory(this.meta.folder)).text, 'director');
1212
1336
 
1213
1337
  try {
@@ -1362,22 +1486,57 @@ export class MissionRun {
1362
1486
  await this.ignoreLocalSettings();
1363
1487
  if (this.meta.status === 'running') this.meta.status = 'interrupted';
1364
1488
 
1365
- // A director's exit is not proof its mission succeeded. The charter makes
1366
- // it write DONE WHEN criteria and tick each one the moment it is actually
1367
- // verified, so criteria still unticked at exit are the director's own
1368
- // record that the work is unfinished — and reporting that as 'done' is
1369
- // the one lie a mission runner cannot afford. Downgrading to
1370
- // 'interrupted' is also the useful answer: it is what makes the run
1371
- // resumable rather than closed.
1372
- // A mission whose own record says every criterion is verified is done,
1373
- // even if the cap ended the turn it was writing its report in. Calling
1374
- // that 'interrupted' told the fleet a finished mission had failed, and
1375
- // invited a resume that spent more to rewrite a report already on disk.
1376
- // The stop reason stays on the record, so nothing is hidden.
1377
- if (this.meta.status === 'interrupted' && this.budgetStopped
1378
- && !this.wasInterrupted && !this.usageLimited) {
1379
- const unmet = await this.unmetCriteria();
1380
- if (doneAtCap({ budgetStopped: this.budgetStopped, wasInterrupted: this.wasInterrupted, usageLimited: this.usageLimited }, unmet)) {
1489
+ await this.settleFinalStatus();
1490
+ this.meta.endedAt = Date.now();
1491
+ this.saveMeta(this.meta);
1492
+ this.emit('run_finished', {
1493
+ status: this.meta.status,
1494
+ costUsd: this.meta.costUsd,
1495
+ directorSessionId: this.meta.directorSessionId,
1496
+ });
1497
+ }
1498
+ }
1499
+
1500
+ /**
1501
+ * The last word on whether this run counts as done — the two records that
1502
+ * outrank the director's own exit, applied in one place.
1503
+ *
1504
+ * A director's exit is not proof its mission succeeded. The charter makes
1505
+ * it write DONE WHEN criteria and tick each one the moment it is actually
1506
+ * verified, so criteria still unticked at exit are the director's own
1507
+ * record that the work is unfinished — and reporting that as 'done' is
1508
+ * the one lie a mission runner cannot afford. Downgrading to
1509
+ * 'interrupted' is also the useful answer: it is what makes the run
1510
+ * resumable rather than closed.
1511
+ * A mission whose own record says every criterion is verified is done,
1512
+ * even if the cap ended the turn it was writing its report in. Calling
1513
+ * that 'interrupted' told the fleet a finished mission had failed, and
1514
+ * invited a resume that spent more to rewrite a report already on disk.
1515
+ * The stop reason stays on the record, so nothing is hidden.
1516
+ *
1517
+ * Separated from `start`'s `finally` because it is the rule, not the
1518
+ * teardown: a test can drive it on a run whose status and crew are set by
1519
+ * hand, which is the only way the gate below is provable without an SDK.
1520
+ */
1521
+ private async settleFinalStatus(): Promise<void> {
1522
+ // The reviewer gate, computed once for the whole method: whether every
1523
+ // reviewer the human marked required has passed the diff this run
1524
+ // actually ends with. Null means nothing is outstanding — either no
1525
+ // required reviewer, or all of them passed the current diff. It gates
1526
+ // both routes to 'done' below, because a run that reached its cap with
1527
+ // every box ticked is in exactly the position the gate exists for: the
1528
+ // director says it is finished and nobody else has looked.
1529
+ const gate = await this.reviewGate();
1530
+ if (this.meta.status === 'interrupted' && this.budgetStopped
1531
+ && !this.wasInterrupted && !this.usageLimited) {
1532
+ const unmet = await this.unmetCriteria();
1533
+ if (doneAtCap({ budgetStopped: this.budgetStopped, wasInterrupted: this.wasInterrupted, usageLimited: this.usageLimited }, unmet)) {
1534
+ if (gate) {
1535
+ // Ticked every box and ran out of budget, but the reviewer the
1536
+ // human required never passed this diff: the run stays interrupted
1537
+ // and resumable, and the event says which reviewer and why.
1538
+ this.emit('mission_unreviewed', { blockers: gate.blockers, text: gate.text });
1539
+ } else {
1381
1540
  this.meta.status = 'done';
1382
1541
  this.emit('mission_done_at_cap', {
1383
1542
  costUsd: this.meta.costUsd, budgetUsd: this.meta.budgetUsd, reason: this.meta.stopReason,
@@ -1386,25 +1545,24 @@ export class MissionRun {
1386
1545
  });
1387
1546
  }
1388
1547
  }
1389
- if (this.meta.status === 'done') {
1390
- const unmet = await this.unmetCriteria();
1391
- if (unmet?.length) {
1392
- this.meta.status = 'interrupted';
1393
- this.emit('mission_incomplete', {
1394
- unmet,
1395
- text: `The director ended with ${unmet.length} DONE WHEN criteri` +
1396
- `${unmet.length === 1 ? 'on' : 'a'} still unticked, so this run is not done. ` +
1397
- 'Resume to continue it.',
1398
- });
1399
- }
1548
+ }
1549
+ if (this.meta.status === 'done') {
1550
+ const unmet = await this.unmetCriteria();
1551
+ if (unmet?.length) {
1552
+ this.meta.status = 'interrupted';
1553
+ this.emit('mission_incomplete', {
1554
+ unmet,
1555
+ text: `The director ended with ${unmet.length} DONE WHEN criteri` +
1556
+ `${unmet.length === 1 ? 'on' : 'a'} still unticked, so this run is not done. ` +
1557
+ 'Resume to continue it.',
1558
+ });
1559
+ } else if (gate) {
1560
+ // Every box ticked and the director says it is finished — and the
1561
+ // reviewer the human required has not passed the diff it ends with.
1562
+ // Foreman refuses the label rather than asking the director to.
1563
+ this.meta.status = 'interrupted';
1564
+ this.emit('mission_unreviewed', { blockers: gate.blockers, text: gate.text });
1400
1565
  }
1401
- this.meta.endedAt = Date.now();
1402
- this.saveMeta(this.meta);
1403
- this.emit('run_finished', {
1404
- status: this.meta.status,
1405
- costUsd: this.meta.costUsd,
1406
- directorSessionId: this.meta.directorSessionId,
1407
- });
1408
1566
  }
1409
1567
  }
1410
1568
 
@@ -1443,19 +1601,13 @@ export class MissionRun {
1443
1601
  * whose director never wrote a doc has already failed more visibly.
1444
1602
  */
1445
1603
  private async unmetCriteria(): Promise<string[] | null> {
1446
- const doc = await readFile(path.join(this.meta.folder, '.foreman', 'MISSION.md'), 'utf8')
1447
- .catch(() => null);
1448
- if (!doc) return null;
1449
-
1450
- const lines = doc.split('\n');
1451
- const start = lines.findIndex((l) => /^#{1,6}\s*DONE\s*WHEN/i.test(l.trim()));
1452
- if (start === -1) return null;
1604
+ const doc = await this.missionDoc();
1605
+ const section = doc === null ? null : doneWhenSection(doc);
1606
+ if (section === null) return null;
1453
1607
 
1454
1608
  const unmet: string[] = [];
1455
1609
  let sawAny = false;
1456
- for (const line of lines.slice(start + 1)) {
1457
- // The section ends at the next heading; checkboxes below it are the plan.
1458
- if (/^#{1,6}\s/.test(line)) break;
1610
+ for (const line of section) {
1459
1611
  const box = line.match(/^\s*[-*]\s*\[( |x|X)\]\s*(.*)$/);
1460
1612
  if (!box) continue;
1461
1613
  sawAny = true;
@@ -1464,6 +1616,11 @@ export class MissionRun {
1464
1616
  return sawAny ? unmet : null;
1465
1617
  }
1466
1618
 
1619
+ /** The mission doc as text, or null when the director never wrote one. */
1620
+ private missionDoc(): Promise<string | null> {
1621
+ return readFile(path.join(this.meta.folder, '.foreman', 'MISSION.md'), 'utf8').catch(() => null);
1622
+ }
1623
+
1467
1624
  /**
1468
1625
  * A headless Playwright browser (its own profile — never the user's
1469
1626
  * Chrome), granted when the mission enabled browser tools.
@@ -1483,7 +1640,14 @@ export class MissionRun {
1483
1640
  };
1484
1641
  }
1485
1642
 
1486
- private policyFor(agent: string) {
1643
+ /**
1644
+ * The permission callback for one agent. `override` tightens (or loosens)
1645
+ * the run's tool policy for that agent alone — see
1646
+ * {@link WorkerOverrides.toolPolicy}. Merged OVER the run's policy rather
1647
+ * than replacing it, so a reviewer still inherits everything the run
1648
+ * decided and only differs where the preset says it must.
1649
+ */
1650
+ private policyFor(agent: string, override?: ToolPolicy) {
1487
1651
  return makePolicy(agent, this.meta.folder, this.runAllowed, this.allowedRoots, {
1488
1652
  onAutoAllow: (a, toolName, reason) => this.emit('auto_allowed', { agent: a, toolName, reason }),
1489
1653
  onAutoDeny: (a, toolName, reason) => this.emit('auto_denied', { agent: a, toolName, reason }),
@@ -1514,7 +1678,10 @@ export class MissionRun {
1514
1678
  if (existed) this.emit('permission_resolved', { id, behavior: 'aborted' });
1515
1679
  return existed;
1516
1680
  },
1517
- }, { toolPolicy: this.meta.toolPolicy, autoAllowReadOnly: this.meta.autoAllowReadOnly });
1681
+ }, {
1682
+ toolPolicy: override ? { ...this.meta.toolPolicy, ...override } : this.meta.toolPolicy,
1683
+ autoAllowReadOnly: this.meta.autoAllowReadOnly,
1684
+ });
1518
1685
  }
1519
1686
 
1520
1687
  /**
@@ -1569,8 +1736,19 @@ export class MissionRun {
1569
1736
  this.meta.costParts = { ...p };
1570
1737
  }
1571
1738
 
1572
- private addCost(usd: number | undefined, role: AgentRole = 'director'): void {
1739
+ private addCost(usd: number | undefined, role: AgentRole = 'director', native = false): void {
1573
1740
  if (typeof usd !== 'number') return;
1741
+ // `native` is for an agent that runs somewhere else entirely — a crew
1742
+ // preset on its own priced provider — where the role's treatment is the
1743
+ // wrong answer and the SDK's dollars are a fact about this run.
1744
+ if (native) {
1745
+ this.costParts.native += usd;
1746
+ this.recomputeCost();
1747
+ this.saveMeta(this.meta);
1748
+ this.emitEconomics();
1749
+ this.enforceBudget();
1750
+ return;
1751
+ }
1574
1752
  // Two cases where the SDK's dollar figure is not a fact about this run:
1575
1753
  // a role Foreman prices itself, and a role behind a gateway at all. The
1576
1754
  // second is the one that leaked — Anthropic's table applied to 5.8M
@@ -1665,7 +1843,7 @@ export class MissionRun {
1665
1843
  };
1666
1844
  }
1667
1845
 
1668
- private addUsage(raw: unknown, role: AgentRole = 'director'): void {
1846
+ private addUsage(raw: unknown, role: AgentRole = 'director', priced = false): void {
1669
1847
  // The real number for this turn just arrived; the estimate standing in
1670
1848
  // for it is now redundant, and the ledger must be re-baselined so the
1671
1849
  // same tokens are not offered again as growth.
@@ -1677,7 +1855,13 @@ export class MissionRun {
1677
1855
  // Where the endpoint published rates, this is the run's real cost: its
1678
1856
  // own tokens at its own prices, accumulated per role so a mixed run bills
1679
1857
  // each half correctly instead of applying one table to both.
1680
- const price = this.prices[role];
1858
+ // `priced` means this agent runs on a provider of its own: either an
1859
+ // Anthropic-native one whose dollars addCost has already taken, or a
1860
+ // gateway that reports through the run's ledger. Either way the ROLE's
1861
+ // rate card is the wrong table — it belongs to a different endpoint — and
1862
+ // applying it would charge tokens that are already accounted for. The
1863
+ // tokens themselves still count.
1864
+ const price = priced ? undefined : this.prices[role];
1681
1865
  if (price) { this.costParts.rated += priceUsage(price, delta); this.recomputeCost(); }
1682
1866
  this.saveMeta(this.meta);
1683
1867
  this.emitEconomics();
@@ -1745,6 +1929,31 @@ export class MissionRun {
1745
1929
  }
1746
1930
 
1747
1931
  /** The mission's branch, when it has one: stay on it, and leave merging and pushing alone. */
1932
+ /**
1933
+ * Who is on this run's crew, by id, and which of them must pass before the
1934
+ * mission can be recorded as done.
1935
+ *
1936
+ * Without this the charter asks for a review the director has no way to
1937
+ * name: the ids live in `meta.crew`, which no prompt showed, so the only
1938
+ * route to them was calling the tool with a wrong id and reading the error.
1939
+ * A gate the director cannot see is a gate it fails by accident.
1940
+ */
1941
+ private crewLine(): string {
1942
+ const crew = this.meta.crew ?? [];
1943
+ if (!crew.length) return '';
1944
+ const rows = crew.map((p) => ` - ${p.id} — ${p.name}${p.model ? ` (${p.model})` : ''}: `
1945
+ + (p.requiredForDone
1946
+ ? 'REQUIRED. This run cannot be recorded as done until it returns VERDICT: PASS on the work as it finally stands.'
1947
+ : 'optional; ask for it when its subject is in play.')).join('\n');
1948
+ const required = crew.filter((p) => p.requiredForDone);
1949
+ return `\n\nYOUR CREW — call mcp__foreman__request_review with one of these ids:\n${rows}\n`
1950
+ + (required.length
1951
+ ? 'Request the required review AFTER your last change and before you tick the final box: a PASS is '
1952
+ + 'pinned to the work as it was reviewed, so anything you change afterwards makes it stale and the '
1953
+ + 'run reads as not done. If a review comes back FAIL, fix what it found and request it again.\n'
1954
+ : '');
1955
+ }
1956
+
1748
1957
  private gitLine(): string {
1749
1958
  const g = this.meta.git;
1750
1959
  if (!g) return '';
@@ -1983,6 +2192,14 @@ export class MissionRun {
1983
2192
  const nextId = `worker-${++this.workerSeq}`;
1984
2193
  this.launchWorker(nextId, prompt, undefined, {
1985
2194
  env: this.agentEnv.director, model: this.meta.directorModel || undefined, priceRole: 'director',
2195
+ // A reviewer retried elsewhere is still a reviewer. Changing which
2196
+ // provider serves it must not hand it Write, Edit and Bash: the
2197
+ // restriction belongs to the job, not to the endpoint.
2198
+ toolPolicy: this.workers.get(workerId)?.overrides?.toolPolicy,
2199
+ // Including the browser: a reviewer retried elsewhere must not gain
2200
+ // clicks and injected script by changing endpoint.
2201
+ noBrowser: this.workers.get(workerId)?.overrides?.noBrowser,
2202
+ crewPresetId: this.workers.get(workerId)?.crewPresetId,
1986
2203
  reason: `retry of ${workerId} on the director's provider, at the human's request`,
1987
2204
  });
1988
2205
  return {
@@ -2025,6 +2242,10 @@ export class MissionRun {
2025
2242
  w.isError = undefined;
2026
2243
  w.reportShown = false;
2027
2244
  w.toolCalls ??= 0;
2245
+ // Kept for the resume path: see WorkerRuntime.overrides. A resume passes
2246
+ // these back in, so `overrides ?? w.overrides` is what a follow-up uses.
2247
+ if (overrides) w.overrides = overrides;
2248
+ if (overrides?.crewPresetId) w.crewPresetId = overrides.crewPresetId;
2028
2249
  w.recent ??= [];
2029
2250
  w.done = new Promise<void>((resolve) => { w.settle = resolve; });
2030
2251
  this.workers.set(workerId, w);
@@ -2079,8 +2300,8 @@ export class MissionRun {
2079
2300
  // Built per worker: the report_progress handler closes over this id,
2080
2301
  // which is how a report lands on the right record without the worker
2081
2302
  // having to know its own name.
2082
- mcpServers: { foreman: this.workerTools(workerId), ...this.browserServers() },
2083
- canUseTool: this.policyFor(workerId),
2303
+ mcpServers: { foreman: this.workerTools(workerId), ...(overrides?.noBrowser ? {} : this.browserServers()) },
2304
+ canUseTool: this.policyFor(workerId, overrides?.toolPolicy),
2084
2305
  },
2085
2306
  });
2086
2307
  w.q = q;
@@ -2138,8 +2359,9 @@ export class MissionRun {
2138
2359
  // usage, so usage has to be current before it is emitted.
2139
2360
  // A worker retried on the director's provider is priced with the
2140
2361
  // director's rates: the tokens went through that gateway.
2141
- this.addUsage(m.usage, overrides?.priceRole ?? 'worker');
2142
- this.addCost(m.total_cost_usd as number | undefined, overrides?.priceRole ?? 'worker');
2362
+ this.addUsage(m.usage, overrides?.priceRole ?? 'worker',
2363
+ overrides?.nativeCost === true || overrides?.ownProvider === true);
2364
+ this.addCost(m.total_cost_usd as number | undefined, overrides?.priceRole ?? 'worker', overrides?.nativeCost === true);
2143
2365
  w.costUsd += (m.total_cost_usd as number | undefined) ?? 0;
2144
2366
  }
2145
2367
  this.emit('message', { agent: workerId, msg });
@@ -2304,6 +2526,210 @@ export class MissionRun {
2304
2526
  `to watch it, and wait_for_worker when you need its result.`;
2305
2527
  }
2306
2528
 
2529
+ /**
2530
+ * request_review — hand the run's diff to a crew reviewer and wait for its
2531
+ * verdict.
2532
+ *
2533
+ * Unlike spawn_worker this AWAITS: the director asked a question and the
2534
+ * answer is the whole point of the call. The reviewer is an ordinary worker
2535
+ * in every other respect — same id sequence, same budget, same transcript —
2536
+ * because a review that did not count against the run's spend would be a
2537
+ * cost nobody could see, and a reviewer with a second kind of id would make
2538
+ * every other surface learn a shape it does not need.
2539
+ *
2540
+ * What it is NOT is a way to ask nicely. The verdict is recorded on the run
2541
+ * whatever the director does with it, and the gate in the end-of-run
2542
+ * `finally` reads that record, not this conversation.
2543
+ */
2544
+ /**
2545
+ * The launch overrides for a worker that is a crew preset, rebuilt from the
2546
+ * run's frozen crew. Used when the record survived a resume but the live
2547
+ * overrides did not: they hold a credential and are deliberately not
2548
+ * persisted, while the restriction they carry must not lapse.
2549
+ */
2550
+ private overridesForPreset(presetId?: string): WorkerOverrides | undefined {
2551
+ if (!presetId) return undefined;
2552
+ const preset = (this.meta.crew ?? []).find((p) => p.id === presetId);
2553
+ if (!preset) return undefined;
2554
+ return {
2555
+ model: preset.model || this.meta.workerModel,
2556
+ env: this.agentEnv.crew?.[preset.id],
2557
+ toolPolicy: preset.toolPolicy === 'default' ? undefined : REVIEWER_TOOL_POLICY,
2558
+ nativeCost: this.roleBasis?.crew?.[preset.id]?.native === true
2559
+ && this.roleBasis.crew[preset.id].basis === 'priced'
2560
+ && Boolean(this.agentEnv.crew?.[preset.id]),
2561
+ noBrowser: preset.toolPolicy !== 'default',
2562
+ ownProvider: Boolean(this.agentEnv.crew?.[preset.id]),
2563
+ crewPresetId: preset.id,
2564
+ reason: `follow-up to ${preset.name} (${preset.id})`,
2565
+ };
2566
+ }
2567
+
2568
+ private async requestReviewTool({ presetId, notes }: { presetId: string; notes?: string }): Promise<string> {
2569
+ const crew = this.meta.crew ?? [];
2570
+ if (!crew.length) {
2571
+ return 'This run has no crew: nobody was added as a reviewer when it was dispatched, so ' +
2572
+ 'there is nothing to request a review from. Verify the work yourself and say so in your report.';
2573
+ }
2574
+ const preset = crew.find((p) => p.id === presetId);
2575
+ if (!preset) {
2576
+ return `No crew preset "${presetId}" on this run. Available: ${crew.map((p) => p.id).join(', ')}.`;
2577
+ }
2578
+ const stop = this.overBudget();
2579
+ if (stop) return stop;
2580
+
2581
+ // The diff as Foreman sees it — the same deck the gate will hash — so a
2582
+ // PASS is about the change Foreman will check, not about whatever the
2583
+ // reviewer happened to look at.
2584
+ // What the reviewer READS is the deck, which is a capped view. What the
2585
+ // PASS is PINNED TO is the working tree itself — see changeFingerprint.
2586
+ // A review that cannot be pinned is not worth having: it would clear the
2587
+ // gate for a diff nobody can prove was the one read.
2588
+ const fingerprint = await changeFingerprint(this.meta.folder);
2589
+ if (fingerprint === null) {
2590
+ return 'Foreman cannot read what this run has changed (the folder is not a git repository, or git ' +
2591
+ 'would not answer), so a review cannot be pinned to it and would not clear the gate. Verify the ' +
2592
+ 'work yourself, say so plainly in your report, and record in MISSION.md that no review was possible.';
2593
+ }
2594
+ const deck = await deckFor(this.meta.folder, this.meta.id);
2595
+ if (deck.baseline?.kind === 'none') {
2596
+ return 'Foreman has no baseline for this run, so it cannot show a reviewer what changed — the deck ' +
2597
+ 'is empty whatever the crew did. A PASS on nothing would clear the gate on nothing, so no review ' +
2598
+ 'is requested. Say this in your report.';
2599
+ }
2600
+ const { diff, truncated } = renderDeckDiff(deck);
2601
+ const doc = await this.missionDoc();
2602
+ const doneWhen = (doc === null ? null : doneWhenSection(doc))?.join('\n') ?? '';
2603
+ let prompt = reviewBriefFor(preset, { mission: this.meta.mission, doneWhen, diff, truncated });
2604
+ if (notes?.trim()) prompt += `\n\n## The director asks you to pay particular attention to\n\n${notes.trim()}`;
2605
+
2606
+ // A reviewer on a priced provider spends real money even when the rest of
2607
+ // the run does not, and a dollar cap that is not "live" is not enforced at
2608
+ // all — enforceBudget and capReached both stand down on an unpriced run.
2609
+ // So the basis moves BEFORE the reviewer is launched, and says so, exactly
2610
+ // as the fallback path does when a human retries on the director's.
2611
+ const presetCost = this.agentEnv.crew?.[preset.id] ? this.roleBasis?.crew?.[preset.id] : undefined;
2612
+ if (presetCost) {
2613
+ const nb = combineBasis(costBasisOf(this.meta), presetCost.basis);
2614
+ if (nb !== costBasisOf(this.meta)) {
2615
+ this.meta.costBasis = nb;
2616
+ this.meta.metered = nb === 'priced';
2617
+ this.emit('settings_changed', {
2618
+ changes: [`${preset.name} reviews on its own provider — this run is now ${nb}`
2619
+ + (nb === 'priced' ? ' and the dollar cap is live' : '')],
2620
+ browserTools: Boolean(this.meta.browserTools), budgetUsd: this.meta.budgetUsd,
2621
+ });
2622
+ this.saveMeta(this.meta);
2623
+ this.emitEconomics();
2624
+ }
2625
+ }
2626
+
2627
+ const id = `worker-${++this.workerSeq}`;
2628
+ const w = this.launchWorker(id, prompt, undefined, {
2629
+ model: preset.model || this.meta.workerModel,
2630
+ // The preset's own provider, when dispatch managed to resolve one for it;
2631
+ // undefined leaves runWorker on the worker env. A reviewer is still worth
2632
+ // running on a fallback provider — refusing to review because a provider
2633
+ // was deleted after the crew was chosen would strand the run at the one
2634
+ // step that lets it be called done.
2635
+ //
2636
+ // Its tokens are still priced at the WORKER role's rates, deliberately.
2637
+ // Per-preset pricing would mean a fourth rate card and a fourth gateway
2638
+ // bucket for a handful of turns; a mixed run already resolves to the
2639
+ // dearer basis rather than the cheaper one (see combineBasis in types.ts),
2640
+ // so the error this leaves behind overstates spend, which is the side a
2641
+ // cost cap should be wrong on.
2642
+ env: this.agentEnv.crew?.[preset.id],
2643
+ // A preset with an env of its own is served by that provider, so its
2644
+ // basis — not the worker role's — decides whether its dollars are real.
2645
+ // Only an Anthropic-native endpoint gives the SDK a dollar figure that
2646
+ // IS the bill. A priced gateway publishes rates and reports through the
2647
+ // shared ledger, so counting the SDK's number there as well would charge
2648
+ // the review twice — the very thing the last fix was about.
2649
+ nativeCost: presetCost?.native === true && presetCost.basis === 'priced',
2650
+ // Its tokens go through its own endpoint, so the worker's rate card
2651
+ // does not describe them whatever the wire is.
2652
+ ownProvider: Boolean(this.agentEnv.crew?.[preset.id]),
2653
+ // A reviewer that must not write must not drive the browser either:
2654
+ // clicks, typing and injected script are automatically permitted once
2655
+ // browser tools are on, so a read-only verdict could change the app it
2656
+ // is judging. The restriction is about the job, not about the file system.
2657
+ noBrowser: preset.toolPolicy !== 'default',
2658
+ // 'default' is the preset saying this role needs to run things; anything
2659
+ // else gets the flat deny, which is what makes "read-only" provable.
2660
+ toolPolicy: preset.toolPolicy === 'default' ? undefined : REVIEWER_TOOL_POLICY,
2661
+ crewPresetId: preset.id,
2662
+ reason: `review by ${preset.name} (${preset.id})`,
2663
+ });
2664
+ const outcome = await w.promise;
2665
+
2666
+ const parsed = parseVerdict(outcome?.report ?? '');
2667
+ const verdict: ReviewVerdict = {
2668
+ presetId: preset.id,
2669
+ name: preset.name,
2670
+ // No verdict line is not a pass. The reviewer may have run out of turns
2671
+ // or answered in prose; either way nobody has said this diff is good,
2672
+ // and the gate must not treat silence as consent.
2673
+ pass: parsed?.pass ?? false,
2674
+ findings: parsed?.findings
2675
+ ?? `${preset.name} never stated a verdict. Its report ended without the required ` +
2676
+ `\`VERDICT: PASS\`/\`VERDICT: FAIL\` line, so this counts as a FAIL.` +
2677
+ (outcome?.report ? `\n\nWhat it did report:\n${outcome.report}` : ''),
2678
+ diffHash: fingerprint,
2679
+ workerId: id,
2680
+ costUsd: this.workers.get(id)?.costUsd,
2681
+ at: Date.now(),
2682
+ };
2683
+ this.meta.reviews = [...(this.meta.reviews ?? []), verdict];
2684
+ this.saveMeta(this.meta);
2685
+ this.emit('review_verdict', {
2686
+ presetId: verdict.presetId, name: verdict.name, model: preset.model || this.meta.workerModel,
2687
+ pass: verdict.pass, findings: verdict.findings, diffHash: verdict.diffHash,
2688
+ workerId: verdict.workerId, costUsd: verdict.costUsd,
2689
+ });
2690
+
2691
+ const head = parsed === null
2692
+ ? `[${preset.name} (${id}) returned NO VERDICT — recorded as a FAIL]`
2693
+ : `[${preset.name} (${id}) — VERDICT: ${verdict.pass ? 'PASS' : 'FAIL'}]`;
2694
+ const tail = verdict.pass
2695
+ ? 'This PASS is pinned to the diff as it stands right now. If you change anything after ' +
2696
+ 'this, the PASS goes stale and the run will not be recorded done — so review last.'
2697
+ : 'This review must pass before the run can be recorded done. Fix what it names, then call ' +
2698
+ 'request_review again.';
2699
+ return `${head}\n${verdict.findings || '(no findings given)'}\n\n${tail}${this.costFooter()}`;
2700
+ }
2701
+
2702
+ /**
2703
+ * THE GATE, as the end of the run sees it: null when nothing required is
2704
+ * outstanding, otherwise the blockers and the sentence for the event.
2705
+ *
2706
+ * A deck that cannot be computed leaves the gate UNSATISFIED rather than
2707
+ * open. "We could not work out what this run changed" is not evidence that
2708
+ * what it changed was reviewed, and the failure mode of the other reading —
2709
+ * a diff error quietly certifying a run as done — is the one this whole
2710
+ * mechanism exists to prevent.
2711
+ */
2712
+ private async reviewGate(): Promise<{ blockers: ReviewBlocker[]; text: string } | null> {
2713
+ const required = (this.meta.crew ?? []).filter((p) => p.requiredForDone);
2714
+ if (!required.length) return null;
2715
+ // deckFor never throws: a missing baseline comes back as a deck with
2716
+ // `baseline.kind === 'none'` and no files, which would hash like a clean
2717
+ // tree and let any PASS stand. So the gate asks git directly, and treats
2718
+ // "cannot tell" as "not reviewed" rather than as "unchanged".
2719
+ const fingerprint = await changeFingerprint(this.meta.folder);
2720
+ if (fingerprint === null) {
2721
+ return {
2722
+ blockers: required.map((p) => ({ presetId: p.id, name: p.name, reason: 'missing' as const })),
2723
+ text: `This run's diff could not be computed, so no review can be matched to it. ` +
2724
+ `${required.map((p) => p.name).join(', ')} ` +
2725
+ `${required.length === 1 ? 'is' : 'are'} required to pass before this run is done, and ` +
2726
+ 'nothing here shows that happened. This run is not done. Resume to continue it.',
2727
+ };
2728
+ }
2729
+ const blockers = reviewBlockers(this.meta.crew, this.meta.reviews, fingerprint);
2730
+ return blockers.length ? { blockers, text: unreviewedText(blockers) } : null;
2731
+ }
2732
+
2307
2733
  private async messageWorkerTool({ worker_id, message }: { worker_id: string; message: string }): Promise<string> {
2308
2734
  const stop = this.overBudget();
2309
2735
  if (stop) return stop;
@@ -2314,7 +2740,12 @@ export class MissionRun {
2314
2740
  if (w.status === 'running') {
2315
2741
  return `${worker_id} is still running — wait_for_worker it first, then send the follow-up.`;
2316
2742
  }
2317
- const run = this.launchWorker(worker_id, message, w.sessionId);
2743
+ // A reviewer's read-only policy, model and provider live in the overrides
2744
+ // its first launch was given. Resuming without them would hand the
2745
+ // director a worker that may now write, on whatever model the run's
2746
+ // workers use — the restriction would quietly expire at the first
2747
+ // follow-up question.
2748
+ const run = this.launchWorker(worker_id, message, w.sessionId, w.overrides ?? this.overridesForPreset(w.crewPresetId));
2318
2749
  await run.promise;
2319
2750
  return `${this.reportLine(run)}${this.costFooter()}`;
2320
2751
  }
@@ -2386,6 +2817,20 @@ export class MissionRun {
2386
2817
  async (args) => text(this.spawnWorkerTool(args)),
2387
2818
  );
2388
2819
 
2820
+ const requestReview = tool(
2821
+ 'request_review',
2822
+ 'Hand this run\'s diff to one of the crew\'s reviewers and get its verdict. Unlike ' +
2823
+ 'spawn_worker this WAITS and returns the findings. The reviewer reads only — it cannot ' +
2824
+ 'change your code — and its verdict is recorded on the run: a reviewer marked required ' +
2825
+ 'must return PASS on the FINAL diff before Foreman will record the run as done, so call ' +
2826
+ 'this after your last change, not before.',
2827
+ {
2828
+ presetId: z.string().describe('The crew preset id to review, e.g. "reviewer"'),
2829
+ notes: z.string().optional().describe('Anything specific you want it to look at, beyond its standing brief'),
2830
+ },
2831
+ async (args) => text(await this.requestReviewTool(args)),
2832
+ );
2833
+
2389
2834
  const checkWorkers = tool(
2390
2835
  'check_workers',
2391
2836
  'Show every worker (or one): status, age, seconds since its last activity, tool calls ' +
@@ -2495,7 +2940,7 @@ export class MissionRun {
2495
2940
  );
2496
2941
  return createSdkMcpServer({
2497
2942
  name: 'foreman',
2498
- tools: [spawnWorker, checkWorkers, waitForWorker, messageWorker, askHuman, exposeService, remember],
2943
+ tools: [spawnWorker, requestReview, checkWorkers, waitForWorker, messageWorker, askHuman, exposeService, remember],
2499
2944
  });
2500
2945
  }
2501
2946
  }