@amenophis1er/foreman 0.1.17 → 0.1.18

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,99 @@
1
+ /**
2
+ * The run side of crew presets: what dispatch freezes onto a record, and what
3
+ * a report says about the verdicts it collected.
4
+ *
5
+ * The freeze test is the one that matters. It runs the server's own pipeline —
6
+ * a settings blob, the overlay, the chosen ids — and then does the thing the
7
+ * whole design exists to survive: it edits the preset afterwards.
8
+ */
9
+ import { test } from 'node:test';
10
+ import assert from 'node:assert/strict';
11
+ import { crewPresetsFrom, type CrewPreset, type ReviewVerdict } from './crew.js';
12
+ import { frozenCrewFor, reviewReportLines } from './run-crew.js';
13
+
14
+ /** What store.readSettings() hands effectiveSettings(), for one project. */
15
+ function settings(): { global: Record<string, unknown>; project: Record<string, unknown> } {
16
+ return {
17
+ global: {
18
+ crewPresets: [
19
+ { id: 'reviewer', name: 'Reviewer', kind: 'reviewer', brief: 'Review it.', requiredForDone: true, model: 'opus' },
20
+ { id: 'perf', name: 'Performance', kind: 'reviewer', brief: 'Look at the hot paths.', requiredForDone: false },
21
+ ],
22
+ },
23
+ project: {},
24
+ };
25
+ }
26
+
27
+ test('the freeze holds: editing a preset afterwards does not change a past run', () => {
28
+ const s = settings();
29
+ // What the server does at dispatch, in the order it does it.
30
+ const presets = crewPresetsFrom(s.global, s.project);
31
+ const run: { crew?: CrewPreset[] } = {};
32
+ const crew = frozenCrewFor(presets, ['reviewer']);
33
+ if (crew) run.crew = crew;
34
+ assert.deepEqual(run.crew, [{ id: 'reviewer', name: 'Reviewer', kind: 'reviewer', brief: 'Review it.', requiredForDone: true, model: 'opus' }]);
35
+
36
+ // Next week, the human edits the preset in Settings: a different brief, and
37
+ // the reviewer no longer blocks a run.
38
+ const source = (s.global.crewPresets as Array<Record<string, unknown>>)[0];
39
+ source.brief = 'Something else entirely.';
40
+ source.requiredForDone = false;
41
+ source.name = 'Renamed';
42
+ // And the list handed out after the edit is a different list, mutated too.
43
+ for (const p of presets) { p.brief = 'mutated'; p.requiredForDone = false; }
44
+
45
+ assert.equal(run.crew![0].brief, 'Review it.', 'the run keeps the brief it was reviewed against');
46
+ assert.equal(run.crew![0].requiredForDone, true, 'and the gate it started under');
47
+ assert.equal(run.crew![0].name, 'Reviewer');
48
+ });
49
+
50
+ test('crewPresetsFrom overlay: the project replaces the global list whole', () => {
51
+ const s = settings();
52
+ s.project.crewPresets = [{ id: 'local', name: 'House reviewer', kind: 'reviewer', brief: 'ours', requiredForDone: true }];
53
+ assert.deepEqual(crewPresetsFrom(s.global, s.project).map((p) => p.id), ['local']);
54
+ assert.deepEqual(crewPresetsFrom(s.global, {}).map((p) => p.id), ['reviewer', 'perf']);
55
+ // The human emptied the project's list: no crew here, not the global one back.
56
+ assert.deepEqual(crewPresetsFrom(s.global, { crewPresets: [] }), []);
57
+ });
58
+
59
+ test('frozenCrewFor: nothing chosen leaves the field absent', () => {
60
+ const presets = crewPresetsFrom(settings().global, {});
61
+ assert.equal(frozenCrewFor(presets, undefined), undefined);
62
+ assert.equal(frozenCrewFor(presets, []), undefined);
63
+ // Only ids nobody has a preset for: still nothing to freeze.
64
+ assert.equal(frozenCrewFor(presets, ['ghost', ' ']), undefined);
65
+ assert.deepEqual(frozenCrewFor(presets, ['perf', 'ghost'])?.map((p) => p.id), ['perf']);
66
+ });
67
+
68
+ const verdict = (over: Partial<ReviewVerdict>): ReviewVerdict => ({
69
+ presetId: 'reviewer', name: 'Reviewer', pass: true, findings: '', diffHash: 'h1', workerId: 'w1', at: 1, ...over,
70
+ });
71
+ const required = (over: Partial<CrewPreset> = {}): CrewPreset =>
72
+ ({ id: 'reviewer', name: 'Reviewer', kind: 'reviewer', brief: '', requiredForDone: true, ...over });
73
+
74
+ test('reviewReportLines: the verdicts, priced when the run was', () => {
75
+ assert.deepEqual(reviewReportLines([required()], [verdict({ costUsd: 0.42 })]), ['reviews: Reviewer PASS · $0.42']);
76
+ assert.deepEqual(
77
+ reviewReportLines([required()], [verdict({ pass: false })]),
78
+ ['reviews: Reviewer FAIL', ' Reviewer is required for this run to be done and returned FAIL.'],
79
+ );
80
+ // Nothing to say about a run that had no crew and collected no verdicts.
81
+ assert.deepEqual(reviewReportLines(undefined, undefined), []);
82
+ });
83
+
84
+ test('reviewReportLines: a required reviewer that never ran is named', () => {
85
+ assert.deepEqual(reviewReportLines([required(), { ...required({ id: 'perf', name: 'Performance' }), requiredForDone: false }], []),
86
+ [' Reviewer is required for this run to be done and has not reviewed it.']);
87
+ });
88
+
89
+ test('reviewReportLines: staleness is claimed only where the diff hash is known', () => {
90
+ const crew = [required()];
91
+ const passed = [verdict({ diffHash: 'old' })];
92
+ // No hash: the report says what the record holds and claims nothing more.
93
+ assert.deepEqual(reviewReportLines(crew, passed), ['reviews: Reviewer PASS']);
94
+ assert.deepEqual(reviewReportLines(crew, passed, 'new'), [
95
+ 'reviews: Reviewer PASS',
96
+ ' Reviewer passed an earlier version of the diff; the code changed after it.',
97
+ ]);
98
+ assert.deepEqual(reviewReportLines(crew, passed, 'old'), ['reviews: Reviewer PASS']);
99
+ });
@@ -0,0 +1,101 @@
1
+ /**
2
+ * The crew as a *run* sees it: what gets frozen onto the record at dispatch,
3
+ * and how the verdicts it collected read back in a report.
4
+ *
5
+ * Both halves are here rather than in crew.ts because crew.ts is the rules —
6
+ * what a preset is, and what blocks a run — while these are the two places
7
+ * Foreman's own surfaces touch them: server.ts freezing at dispatch, and
8
+ * server.ts and mcp.ts writing the same lines to the phone and to MCP. Two
9
+ * surfaces printing verdicts two different ways is how "PASS" comes to mean
10
+ * something slightly different depending on where you read it.
11
+ */
12
+ import { freezeCrew, reviewBlockers, type CrewPreset, type ReviewVerdict } from './crew.js';
13
+
14
+ /**
15
+ * The crew to freeze onto a new run, or undefined when there is none.
16
+ *
17
+ * undefined rather than [] on purpose: an absent field is how every run
18
+ * recorded before presets existed reads, and a run dispatched with no crew is
19
+ * that same run. Unknown ids fall away silently — freezeCrew resolves against
20
+ * the presets actually in force, so an id from a preset the human has since
21
+ * deleted cannot conjure a reviewer that no longer exists.
22
+ */
23
+ export function frozenCrewFor(
24
+ presets: readonly CrewPreset[],
25
+ ids: readonly string[] | undefined,
26
+ ): CrewPreset[] | undefined {
27
+ if (!Array.isArray(ids) || ids.length === 0) return undefined;
28
+ const clean = ids.filter((id): id is string => typeof id === 'string' && id.trim() !== '');
29
+ const crew = freezeCrew(clean, presets);
30
+ return crew.length ? crew : undefined;
31
+ }
32
+
33
+ /**
34
+ * The reviewers whose PASS this finished run is entitled to wear, or [] .
35
+ *
36
+ * Derived here and sent as a name list rather than shipping `reviews` into the
37
+ * fleet's projection, because that projection exists to stay small: it is
38
+ * re-polled every few seconds for every project, and a verdict's findings run
39
+ * to thousands of characters that a tile never renders. The tile asks one
40
+ * question — "did the required reviewers pass?" — so it is handed the answer.
41
+ *
42
+ * Only a run Foreman actually recorded `done` qualifies, and that is what
43
+ * makes staleness answerable from a record alone: the end-of-run gate refuses
44
+ * `done` unless every required PASS was pinned to the diff the run ended with,
45
+ * so a `done` run with all its required verdicts passing was current when it
46
+ * mattered. Anything else gets no mark rather than a mark that might be a lie.
47
+ */
48
+ export function reviewedByNames(run: {
49
+ status: string;
50
+ crew?: readonly CrewPreset[];
51
+ reviews?: readonly ReviewVerdict[];
52
+ }): string[] {
53
+ if (run.status !== 'done') return [];
54
+ const required = (run.crew ?? []).filter((p) => p.requiredForDone);
55
+ if (!required.length) return [];
56
+ const names: string[] = [];
57
+ for (const p of required) {
58
+ const passed = (run.reviews ?? []).filter((v) => v.presetId === p.id && v.pass);
59
+ if (!passed.length) return [];
60
+ names.push(p.name);
61
+ }
62
+ return names;
63
+ }
64
+
65
+ /** `$0.42` when the run's provider priced it, nothing when it did not. */
66
+ function cost(v: ReviewVerdict): string {
67
+ return typeof v.costUsd === 'number' && v.costUsd > 0 ? ` · $${v.costUsd.toFixed(2)}` : '';
68
+ }
69
+
70
+ /**
71
+ * The `reviews:` block for a run report, or [] when the run collected none.
72
+ *
73
+ * `currentDiffHash` is optional because most surfaces cannot honestly compute
74
+ * it: a report reads a frozen deck whose diffs are capped, so hashing it would
75
+ * call a perfectly good PASS stale. Without it, staleness is simply not
76
+ * claimed — the end-of-run gate is the one place that judges it — and the
77
+ * other two blockers, a required reviewer that never ran and one that said
78
+ * FAIL, are facts the record already holds.
79
+ */
80
+ export function reviewReportLines(
81
+ crew: readonly CrewPreset[] | undefined,
82
+ reviews: readonly ReviewVerdict[] | undefined,
83
+ currentDiffHash?: string,
84
+ ): string[] {
85
+ const verdicts = reviews ?? [];
86
+ if (!verdicts.length && !(crew ?? []).length) return [];
87
+ const lines: string[] = [];
88
+ if (verdicts.length) {
89
+ lines.push(`reviews: ${verdicts.map((v) => `${v.name} ${v.pass ? 'PASS' : 'FAIL'}${cost(v)}`).join(', ')}`);
90
+ }
91
+ const blockers = reviewBlockers(crew, verdicts, currentDiffHash ?? '')
92
+ .filter((b) => (currentDiffHash === undefined ? b.reason !== 'stale' : true));
93
+ for (const b of blockers) {
94
+ lines.push(b.reason === 'missing'
95
+ ? ` ${b.name} is required for this run to be done and has not reviewed it.`
96
+ : b.reason === 'fail'
97
+ ? ` ${b.name} is required for this run to be done and returned FAIL.`
98
+ : ` ${b.name} passed an earlier version of the diff; the code changed after it.`);
99
+ }
100
+ return lines;
101
+ }
package/src/server.ts CHANGED
@@ -82,6 +82,8 @@ import { reconcileRole } from './role-provider.js';
82
82
  import { detectBrowser, installChromium } from './browser.js';
83
83
  import { frozenDeck, frozenMissionDoc, parkMissionDoc, restoreMissionDoc, snapshotRun } from './snapshot.js';
84
84
  import { closeMissionBranch, compareUrl, createPullRequest, dirtyPaths, ensureMissionBranch, ghReady, gitInfo, resolvePrBase, worktreeGrant, worktreeParent, missionBranchName, prDraft, pullRequestState, pushBranch, renameMissionBranch, startMissionBranch, type GitInfo } from './gitwork.js';
85
+ import { crewPresetsFrom, type CrewPreset } from './crew.js';
86
+ import { frozenCrewFor, reviewReportLines, reviewedByNames } from './run-crew.js';
85
87
  import { detectTailscale, tailnetUrl } from './tailscale.js';
86
88
  import { checkForUpdate, currentVersion, type UpdateInfo } from './update.js';
87
89
  import { ServiceRegistry, listeningPid, portOpen, servicesHandler, stopService } from './services.js';
@@ -702,6 +704,101 @@ async function agentEnvFor(
702
704
  return providerEnv(resolved, ledgerKey ? `${url}/run/${encodeURIComponent(ledgerKey)}` : url);
703
705
  }
704
706
 
707
+ /**
708
+ * An agent env per crew preset that named a provider of its own, keyed by
709
+ * preset id — the third kind of env a run can need, after its two roles. A
710
+ * reviewer pinned to another provider is a real request: the point of a second
711
+ * opinion is partly that it comes from a different model, and a model carries
712
+ * its provider.
713
+ *
714
+ * Only presets that name a `providerId` get an entry, and only when that
715
+ * provider resolves cleanly. Anything else — no id, an id nothing resolves,
716
+ * a resolution `providerProblem` calls unusable — is LEFT OUT rather than
717
+ * reported, because a missing entry means the reviewer runs on the worker env
718
+ * and a thrown error means the run dies. A provider deleted between the moment
719
+ * a human chose the crew and the moment the director asks for the review must
720
+ * degrade to the worker's provider: the alternative is a mission stranded at
721
+ * the one step that would let it be recorded as done, over a credential its
722
+ * reviewer never strictly needed.
723
+ *
724
+ * Each preset's env carries the preset's own model for the same reason the two
725
+ * roles do — the SDK's aliases have to resolve to something this provider's
726
+ * gateway serves — and shares the run's ledger key, since the reviewer's
727
+ * tokens are accounted to the worker role. See the `env` comment in
728
+ * orchestrator.ts's requestReviewTool for why that approximation is chosen.
729
+ */
730
+ /**
731
+ * Each crew preset's own cost basis, by preset id — for the presets that name
732
+ * a provider Foreman can resolve, which are the ones that will run somewhere
733
+ * other than the worker role.
734
+ *
735
+ * The run needs this to count a review's spend honestly. A reviewer pinned to
736
+ * a paid provider beside workers on a free gateway had its dollars discarded,
737
+ * because cost was attributed to the worker role and a gateway role's figures
738
+ * are dropped by design. Whether money is real is the provider's answer, not
739
+ * the role's.
740
+ */
741
+ async function crewBasesFor(meta: RunMeta): Promise<Record<string, { basis: CostBasis; native: boolean }> | undefined> {
742
+ const wanted = (meta.crew ?? []).filter((p) => p.providerId);
743
+ if (!wanted.length) return undefined;
744
+ const out: Record<string, { basis: CostBasis; native: boolean }> = {};
745
+ const seen = new Map<string, { basis: CostBasis; native: boolean } | null>();
746
+ for (const preset of wanted) {
747
+ const id = preset.providerId as string;
748
+ if (!seen.has(id)) {
749
+ const ref = providerForRole(meta, id);
750
+ const resolved = 'id' in ref && ref.id === id
751
+ ? await resolveProvider(ref, store.root).catch(() => null)
752
+ : null;
753
+ seen.set(id, resolved && !providerProblem(resolved)
754
+ ? {
755
+ basis: (await roleCost(withRoleModel(resolved, preset.model ?? meta.workerModel), preset.model ?? meta.workerModel)).basis,
756
+ // Whether the SDK's dollar figure for this preset IS the bill, or
757
+ // whether it is a gateway that reports through the run's ledger.
758
+ native: resolved.wire === 'anthropic-native',
759
+ }
760
+ : null);
761
+ }
762
+ const cost = seen.get(id);
763
+ if (cost) out[preset.id] = cost;
764
+ }
765
+ return Object.keys(out).length ? out : undefined;
766
+ }
767
+
768
+ async function crewEnvsFor(
769
+ meta: RunMeta, ledgerKey: string,
770
+ ): Promise<Record<string, ReturnType<typeof providerEnv>> | undefined> {
771
+ const wanted = (meta.crew ?? []).filter((p) => p.providerId);
772
+ if (!wanted.length) return undefined;
773
+ // One resolution per distinct provider, not per preset: two reviewers on the
774
+ // same endpoint are one credential and one gateway.
775
+ const bases = new Map<string, ResolvedProvider | null>();
776
+ const out: Record<string, ReturnType<typeof providerEnv>> = {};
777
+ for (const preset of wanted) {
778
+ const id = preset.providerId as string;
779
+ if (!bases.has(id)) {
780
+ const ref = providerForRole(meta, id);
781
+ // providerForRole answers with the RUN's provider when the id resolves to
782
+ // nothing it knows. That is the right answer for a role, which must run
783
+ // somewhere; here it would quietly pin the preset to a provider nobody
784
+ // asked for, so it counts as "no entry" instead.
785
+ const resolved = 'id' in ref && ref.id === id
786
+ ? await resolveProvider(ref, store.root).catch(() => null)
787
+ : null;
788
+ bases.set(id, resolved && !providerProblem(resolved) ? resolved : null);
789
+ }
790
+ const base = bases.get(id);
791
+ if (!base) continue;
792
+ const withModel = withRoleModel(base, preset.model ?? meta.workerModel);
793
+ if (providerProblem(withModel)) continue;
794
+ // A gateway that will not start is the same kind of nothing: the reviewer
795
+ // falls back rather than the run failing.
796
+ const env = await agentEnvFor(withModel, meta.id, ledgerKey).catch(() => null);
797
+ if (env) out[preset.id] = env;
798
+ }
799
+ return Object.keys(out).length ? out : undefined;
800
+ }
801
+
705
802
  /**
706
803
  * The ledger bucket for this attempt at a run.
707
804
  *
@@ -1118,6 +1215,10 @@ const fleetHost: FleetHost = {
1118
1215
  ];
1119
1216
  if (words.error) lines.push(`it stopped with: ${clipText(words.error, 400)}`);
1120
1217
  if (last.workers.length) lines.push(`crew: ${last.workers.map((w) => `${w.id} ${w.status}`).join(', ')}`);
1218
+ // Verdicts read out here and nowhere else on the phone: which reviewers
1219
+ // ran is a fact about the run, while which presets exist is configuration,
1220
+ // and configuration is edited on the dashboard.
1221
+ lines.push(...reviewReportLines(last.crew, last.reviews));
1121
1222
  if (words.result) lines.push(`director's closing report:\n${clipText(words.result, 2500)}`);
1122
1223
  else if (words.last) lines.push(`director's last words:\n${clipText(words.last, 2500)}`);
1123
1224
  return lines.join('\n');
@@ -1631,8 +1732,8 @@ async function driveRun(
1631
1732
  let agentEnv;
1632
1733
  let roleBasis = resolved.costBasis;
1633
1734
  let prices: { director?: ModelPrice; worker?: ModelPrice } = {};
1634
- let roleBases: { director: CostBasis; worker: CostBasis } | undefined;
1635
- let gatewayRoles = { director: false, worker: false };
1735
+ let roleBases: { director: CostBasis; worker: CostBasis; crew?: Record<string, { basis: CostBasis; native: boolean }> } | undefined;
1736
+ let gatewayRoles: { director: boolean; worker: boolean; crew?: boolean } = { director: false, worker: false };
1636
1737
  try {
1637
1738
  // Resolved per role. Where both roles share a provider this resolves once
1638
1739
  // and starts one gateway; where they differ, the supervisor already runs a
@@ -1662,11 +1763,15 @@ async function driveRun(
1662
1763
  worker: directorProvider === workerProvider
1663
1764
  ? await agentEnvFor(directorProvider, meta.id, key)
1664
1765
  : await agentEnvFor(workerProvider, meta.id, key),
1766
+ crew: await crewEnvsFor(meta, key),
1665
1767
  };
1666
1768
  // Only roles that actually go through a gateway are counted there; a
1667
1769
  // native role's tokens arrive on the SDK's own result message, and adding
1668
1770
  // both would double every one of them.
1669
1771
  gatewayRoles = {
1772
+ // A crew preset on a gateway sends its tokens to the same ledger, so
1773
+ // polling has to run even when both ordinary roles are native.
1774
+ crew: Object.values(await crewBasesFor(meta) ?? {}).some((c) => !c.native),
1670
1775
  director: directorProvider.wire !== 'anthropic-native',
1671
1776
  worker: workerProvider.wire !== 'anthropic-native',
1672
1777
  };
@@ -1676,7 +1781,7 @@ async function driveRun(
1676
1781
  : await roleCost(workerProvider, meta.workerModel);
1677
1782
  roleBasis = combineBasis(directorCost.basis, workerCost.basis);
1678
1783
  prices = { director: directorCost.price, worker: workerCost.price };
1679
- roleBases = { director: directorCost.basis, worker: workerCost.basis };
1784
+ roleBases = { director: directorCost.basis, worker: workerCost.basis, crew: await crewBasesFor(meta) };
1680
1785
  } catch (err) {
1681
1786
  meta.status = 'error';
1682
1787
  meta.endedAt = Date.now();
@@ -1799,6 +1904,8 @@ async function effectiveSettings(projectId: string): Promise<{
1799
1904
  budgetWarnAt: number;
1800
1905
  /** Ceiling on what this project's SCHEDULED runs may cost in one calendar month. */
1801
1906
  scheduledMonthlyCapUsd: number;
1907
+ /** The crew presets a mission here may be started with (built-ins until edited). */
1908
+ crewPresets: CrewPreset[];
1802
1909
  }> {
1803
1910
  const s = await store.readSettings()
1804
1911
  .catch(() => ({ global: {}, projects: {} as Record<string, object> }));
@@ -1831,6 +1938,9 @@ async function effectiveSettings(projectId: string): Promise<{
1831
1938
  const raw = Number(p.scheduledMonthlyCapUsd ?? g.scheduledMonthlyCapUsd);
1832
1939
  return Number.isFinite(raw) && raw >= 0 ? raw : DEFAULT_SCHEDULED_MONTHLY_CAP_USD;
1833
1940
  })(),
1941
+ // The project's list replaces the global one whole, like the rest of the
1942
+ // overlay — merging would make "no reviewer on this project" unsayable.
1943
+ crewPresets: crewPresetsFrom(g, p),
1834
1944
  };
1835
1945
  }
1836
1946
 
@@ -1912,6 +2022,12 @@ async function startRun(
1912
2022
  scheduleId?: string;
1913
2023
  scheduleName?: string;
1914
2024
  } = {},
2025
+ /**
2026
+ * Crew presets the human opted this mission into, by id. Resolved and copied
2027
+ * onto the record here and nowhere else: the run is reviewed against the
2028
+ * presets as they stood when it started, whatever Settings says later.
2029
+ */
2030
+ crewIds?: readonly string[],
1915
2031
  ): Promise<void> {
1916
2032
  const settings = await effectiveSettings(projectId);
1917
2033
  if (origin.scheduleId && origin.scheduleName) scheduleNames.set(origin.scheduleId, origin.scheduleName);
@@ -1937,6 +2053,13 @@ async function startRun(
1937
2053
  // must not silently move an in-flight or resumed run to another provider,
1938
2054
  // or another bill.
1939
2055
  provider,
2056
+ // Frozen for the same reason as the provider, and against a stronger
2057
+ // temptation: a preset edited next week must not change what a run that is
2058
+ // still going — or one that finished in March — was reviewed against.
2059
+ ...(() => {
2060
+ const crew = frozenCrewFor(settings.crewPresets, crewIds);
2061
+ return crew ? { crew } : {};
2062
+ })(),
1940
2063
  status: 'running', costUsd: 0,
1941
2064
  createdAt: Date.now(), workers: [],
1942
2065
  };
@@ -2531,6 +2654,10 @@ const server = http.createServer(async (req, res) => {
2531
2654
  createdAt: lastRun.createdAt, costUsd: lastRun.costUsd,
2532
2655
  // The card may print a dollar only where the dollar was real.
2533
2656
  costBasis: costBasisOf(lastRun), usage: lastRun.usage,
2657
+ // Not the verdicts themselves — see reviewedByNames. The tile
2658
+ // renders a glyph and a name, and this projection is polled for
2659
+ // every project every few seconds.
2660
+ reviewedBy: reviewedByNames(lastRun),
2534
2661
  },
2535
2662
  // When this project last did anything, so the fleet can lead with it.
2536
2663
  // A planner parked on a question is doing something — waiting on
@@ -2680,7 +2807,7 @@ const server = http.createServer(async (req, res) => {
2680
2807
  } else if (req.method === 'POST' && url.pathname === '/run') {
2681
2808
  const {
2682
2809
  projectId, mission, budgetUsd, directorModel, workerModel, browserTools,
2683
- directorProviderId, workerProviderId, allowDirty, startedBy,
2810
+ directorProviderId, workerProviderId, allowDirty, startedBy, crew,
2684
2811
  } = await readBody(req);
2685
2812
  if (typeof projectId !== 'string' || typeof mission !== 'string' || !mission.trim()) {
2686
2813
  return json(res, 400, { error: 'projectId and mission are required' });
@@ -2726,7 +2853,10 @@ const server = http.createServer(async (req, res) => {
2726
2853
  // 'schedule' is deliberately not accepted here: a caller must not be
2727
2854
  // able to forge a scheduled start and charge the month's unattended
2728
2855
  // allowance for a run no schedule asked for.
2729
- { startedBy: startedBy === 'phone' || startedBy === 'mcp' ? startedBy : 'human' });
2856
+ { startedBy: startedBy === 'phone' || startedBy === 'mcp' ? startedBy : 'human' },
2857
+ // The composer is the only place a crew is chosen, so it is the only
2858
+ // dispatch path that carries one; unknown ids are dropped downstream.
2859
+ Array.isArray(crew) ? crew as string[] : undefined);
2730
2860
  json(res, 200, { ok: true });
2731
2861
 
2732
2862
  } else if (url.pathname === '/notify' && req.method === 'GET') {
@@ -3195,7 +3325,7 @@ const server = http.createServer(async (req, res) => {
3195
3325
  // mission was branched from — see resolvePrBase.
3196
3326
  const target = await resolvePrBase(meta.folder, meta.git.base);
3197
3327
  json(res, 200, {
3198
- ...prDraft(meta, doc), branch: meta.git.branch, base: target.base, commits: meta.git.commits ?? null,
3328
+ ...prDraft(meta, doc, meta.reviews), branch: meta.git.branch, base: target.base, commits: meta.git.commits ?? null,
3199
3329
  branchedFrom: meta.git.base, baseFellBack: target.fellBack,
3200
3330
  remote: info.remote, compareUrl: compareUrl(info.remote, target.base, meta.git.branch),
3201
3331
  gh, pr: meta.git.pr ?? null, onBranch: info.branch === meta.git.branch, dirty: Boolean(info.dirty),
package/src/types.ts CHANGED
@@ -6,11 +6,17 @@
6
6
  * so replaying a log reproduces exactly what a live client observed.
7
7
  */
8
8
  import type { Cadence } from './schedule.js';
9
+ import type { CrewPreset, ReviewVerdict } from './crew.js';
9
10
 
10
11
  /** Re-exported so callers can name a schedule's cadence without reaching past
11
12
  * this module for it; the rules that interpret one live in schedule.ts. */
12
13
  export type { Cadence };
13
14
 
15
+ /** Re-exported on the same principle: a record can be described without
16
+ * reaching past this module, while crew.ts stays the definition site and
17
+ * keeps the rules that read them. */
18
+ export type { CrewPreset, ReviewVerdict };
19
+
14
20
  export type RunStatus = 'running' | 'done' | 'error' | 'interrupted';
15
21
 
16
22
  /** A linked project — a folder Foreman runs missions in. */
@@ -249,6 +255,14 @@ export interface WorkerMeta {
249
255
  status: WorkerStatus;
250
256
  costUsd: number;
251
257
  sessionId?: string;
258
+ /**
259
+ * The crew preset this worker IS, when it is a reviewer rather than an
260
+ * ordinary worker. Persisted — unlike the launch overrides, which hold a
261
+ * live credential — so a resumed run can rebuild the read-only policy, the
262
+ * model and the provider from the frozen crew instead of resuming a
263
+ * reviewer as a worker that may write.
264
+ */
265
+ crewPresetId?: string;
252
266
  /** First 500 chars of the task brief, for run-history display. */
253
267
  task: string;
254
268
  /**
@@ -353,6 +367,19 @@ export interface RunMeta {
353
367
  provider?: ProviderRef;
354
368
  /** @deprecated Pre-provider pin, still read for runs recorded before providers. */
355
369
  claudeInstance?: ClaudeInstanceRef;
370
+ /**
371
+ * Crew presets frozen onto the run at dispatch: a later edit of a preset
372
+ * must not change a running or past run. Absent means the run was dispatched
373
+ * with no crew chosen — which is every run recorded before presets existed,
374
+ * and the reason the gate reads an absent field as "nothing required".
375
+ */
376
+ crew?: CrewPreset[];
377
+ /**
378
+ * Review verdicts this run collected, appended in order. Kept whole rather
379
+ * than reduced to a pass/fail: the gate needs the diff each verdict was
380
+ * about, and the human reading the record afterwards needs the findings.
381
+ */
382
+ reviews?: ReviewVerdict[];
356
383
  /** Number of times this run was resumed after an interruption. */
357
384
  resumes?: number;
358
385
  /**
@@ -526,6 +553,8 @@ export interface MissionProposal {
526
553
  directorProviderId?: string;
527
554
  workerProviderId?: string;
528
555
  modelRationale?: string;
556
+ /** Preset ids the planner may suggest; the human's toggles decide. */
557
+ crew?: string[];
529
558
  createdAt: number;
530
559
  }
531
560