@amenophis1er/foreman 0.1.17 → 0.1.18
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +32 -1
- package/package.json +1 -1
- package/src/crew.test.ts +376 -0
- package/src/crew.ts +330 -0
- package/src/gitwork.test.ts +84 -2
- package/src/gitwork.ts +143 -2
- package/src/mcp.ts +11 -2
- package/src/orchestrator.test.ts +516 -2
- package/src/orchestrator.ts +514 -69
- package/src/run-crew.test.ts +99 -0
- package/src/run-crew.ts +101 -0
- package/src/server.ts +136 -6
- package/src/types.ts +29 -0
- package/ui/dist/assets/index-0QuGXbFg.js +76 -0
- package/ui/dist/index.html +1 -1
- package/ui/dist/assets/index-Bcc4KMtO.js +0 -68
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The run side of crew presets: what dispatch freezes onto a record, and what
|
|
3
|
+
* a report says about the verdicts it collected.
|
|
4
|
+
*
|
|
5
|
+
* The freeze test is the one that matters. It runs the server's own pipeline —
|
|
6
|
+
* a settings blob, the overlay, the chosen ids — and then does the thing the
|
|
7
|
+
* whole design exists to survive: it edits the preset afterwards.
|
|
8
|
+
*/
|
|
9
|
+
import { test } from 'node:test';
|
|
10
|
+
import assert from 'node:assert/strict';
|
|
11
|
+
import { crewPresetsFrom, type CrewPreset, type ReviewVerdict } from './crew.js';
|
|
12
|
+
import { frozenCrewFor, reviewReportLines } from './run-crew.js';
|
|
13
|
+
|
|
14
|
+
/** What store.readSettings() hands effectiveSettings(), for one project. */
|
|
15
|
+
function settings(): { global: Record<string, unknown>; project: Record<string, unknown> } {
|
|
16
|
+
return {
|
|
17
|
+
global: {
|
|
18
|
+
crewPresets: [
|
|
19
|
+
{ id: 'reviewer', name: 'Reviewer', kind: 'reviewer', brief: 'Review it.', requiredForDone: true, model: 'opus' },
|
|
20
|
+
{ id: 'perf', name: 'Performance', kind: 'reviewer', brief: 'Look at the hot paths.', requiredForDone: false },
|
|
21
|
+
],
|
|
22
|
+
},
|
|
23
|
+
project: {},
|
|
24
|
+
};
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
test('the freeze holds: editing a preset afterwards does not change a past run', () => {
|
|
28
|
+
const s = settings();
|
|
29
|
+
// What the server does at dispatch, in the order it does it.
|
|
30
|
+
const presets = crewPresetsFrom(s.global, s.project);
|
|
31
|
+
const run: { crew?: CrewPreset[] } = {};
|
|
32
|
+
const crew = frozenCrewFor(presets, ['reviewer']);
|
|
33
|
+
if (crew) run.crew = crew;
|
|
34
|
+
assert.deepEqual(run.crew, [{ id: 'reviewer', name: 'Reviewer', kind: 'reviewer', brief: 'Review it.', requiredForDone: true, model: 'opus' }]);
|
|
35
|
+
|
|
36
|
+
// Next week, the human edits the preset in Settings: a different brief, and
|
|
37
|
+
// the reviewer no longer blocks a run.
|
|
38
|
+
const source = (s.global.crewPresets as Array<Record<string, unknown>>)[0];
|
|
39
|
+
source.brief = 'Something else entirely.';
|
|
40
|
+
source.requiredForDone = false;
|
|
41
|
+
source.name = 'Renamed';
|
|
42
|
+
// And the list handed out after the edit is a different list, mutated too.
|
|
43
|
+
for (const p of presets) { p.brief = 'mutated'; p.requiredForDone = false; }
|
|
44
|
+
|
|
45
|
+
assert.equal(run.crew![0].brief, 'Review it.', 'the run keeps the brief it was reviewed against');
|
|
46
|
+
assert.equal(run.crew![0].requiredForDone, true, 'and the gate it started under');
|
|
47
|
+
assert.equal(run.crew![0].name, 'Reviewer');
|
|
48
|
+
});
|
|
49
|
+
|
|
50
|
+
test('crewPresetsFrom overlay: the project replaces the global list whole', () => {
|
|
51
|
+
const s = settings();
|
|
52
|
+
s.project.crewPresets = [{ id: 'local', name: 'House reviewer', kind: 'reviewer', brief: 'ours', requiredForDone: true }];
|
|
53
|
+
assert.deepEqual(crewPresetsFrom(s.global, s.project).map((p) => p.id), ['local']);
|
|
54
|
+
assert.deepEqual(crewPresetsFrom(s.global, {}).map((p) => p.id), ['reviewer', 'perf']);
|
|
55
|
+
// The human emptied the project's list: no crew here, not the global one back.
|
|
56
|
+
assert.deepEqual(crewPresetsFrom(s.global, { crewPresets: [] }), []);
|
|
57
|
+
});
|
|
58
|
+
|
|
59
|
+
test('frozenCrewFor: nothing chosen leaves the field absent', () => {
|
|
60
|
+
const presets = crewPresetsFrom(settings().global, {});
|
|
61
|
+
assert.equal(frozenCrewFor(presets, undefined), undefined);
|
|
62
|
+
assert.equal(frozenCrewFor(presets, []), undefined);
|
|
63
|
+
// Only ids nobody has a preset for: still nothing to freeze.
|
|
64
|
+
assert.equal(frozenCrewFor(presets, ['ghost', ' ']), undefined);
|
|
65
|
+
assert.deepEqual(frozenCrewFor(presets, ['perf', 'ghost'])?.map((p) => p.id), ['perf']);
|
|
66
|
+
});
|
|
67
|
+
|
|
68
|
+
const verdict = (over: Partial<ReviewVerdict>): ReviewVerdict => ({
|
|
69
|
+
presetId: 'reviewer', name: 'Reviewer', pass: true, findings: '', diffHash: 'h1', workerId: 'w1', at: 1, ...over,
|
|
70
|
+
});
|
|
71
|
+
const required = (over: Partial<CrewPreset> = {}): CrewPreset =>
|
|
72
|
+
({ id: 'reviewer', name: 'Reviewer', kind: 'reviewer', brief: '', requiredForDone: true, ...over });
|
|
73
|
+
|
|
74
|
+
test('reviewReportLines: the verdicts, priced when the run was', () => {
|
|
75
|
+
assert.deepEqual(reviewReportLines([required()], [verdict({ costUsd: 0.42 })]), ['reviews: Reviewer PASS · $0.42']);
|
|
76
|
+
assert.deepEqual(
|
|
77
|
+
reviewReportLines([required()], [verdict({ pass: false })]),
|
|
78
|
+
['reviews: Reviewer FAIL', ' Reviewer is required for this run to be done and returned FAIL.'],
|
|
79
|
+
);
|
|
80
|
+
// Nothing to say about a run that had no crew and collected no verdicts.
|
|
81
|
+
assert.deepEqual(reviewReportLines(undefined, undefined), []);
|
|
82
|
+
});
|
|
83
|
+
|
|
84
|
+
test('reviewReportLines: a required reviewer that never ran is named', () => {
|
|
85
|
+
assert.deepEqual(reviewReportLines([required(), { ...required({ id: 'perf', name: 'Performance' }), requiredForDone: false }], []),
|
|
86
|
+
[' Reviewer is required for this run to be done and has not reviewed it.']);
|
|
87
|
+
});
|
|
88
|
+
|
|
89
|
+
test('reviewReportLines: staleness is claimed only where the diff hash is known', () => {
|
|
90
|
+
const crew = [required()];
|
|
91
|
+
const passed = [verdict({ diffHash: 'old' })];
|
|
92
|
+
// No hash: the report says what the record holds and claims nothing more.
|
|
93
|
+
assert.deepEqual(reviewReportLines(crew, passed), ['reviews: Reviewer PASS']);
|
|
94
|
+
assert.deepEqual(reviewReportLines(crew, passed, 'new'), [
|
|
95
|
+
'reviews: Reviewer PASS',
|
|
96
|
+
' Reviewer passed an earlier version of the diff; the code changed after it.',
|
|
97
|
+
]);
|
|
98
|
+
assert.deepEqual(reviewReportLines(crew, passed, 'old'), ['reviews: Reviewer PASS']);
|
|
99
|
+
});
|
package/src/run-crew.ts
ADDED
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The crew as a *run* sees it: what gets frozen onto the record at dispatch,
|
|
3
|
+
* and how the verdicts it collected read back in a report.
|
|
4
|
+
*
|
|
5
|
+
* Both halves are here rather than in crew.ts because crew.ts is the rules —
|
|
6
|
+
* what a preset is, and what blocks a run — while these are the two places
|
|
7
|
+
* Foreman's own surfaces touch them: server.ts freezing at dispatch, and
|
|
8
|
+
* server.ts and mcp.ts writing the same lines to the phone and to MCP. Two
|
|
9
|
+
* surfaces printing verdicts two different ways is how "PASS" comes to mean
|
|
10
|
+
* something slightly different depending on where you read it.
|
|
11
|
+
*/
|
|
12
|
+
import { freezeCrew, reviewBlockers, type CrewPreset, type ReviewVerdict } from './crew.js';
|
|
13
|
+
|
|
14
|
+
/**
|
|
15
|
+
* The crew to freeze onto a new run, or undefined when there is none.
|
|
16
|
+
*
|
|
17
|
+
* undefined rather than [] on purpose: an absent field is how every run
|
|
18
|
+
* recorded before presets existed reads, and a run dispatched with no crew is
|
|
19
|
+
* that same run. Unknown ids fall away silently — freezeCrew resolves against
|
|
20
|
+
* the presets actually in force, so an id from a preset the human has since
|
|
21
|
+
* deleted cannot conjure a reviewer that no longer exists.
|
|
22
|
+
*/
|
|
23
|
+
export function frozenCrewFor(
|
|
24
|
+
presets: readonly CrewPreset[],
|
|
25
|
+
ids: readonly string[] | undefined,
|
|
26
|
+
): CrewPreset[] | undefined {
|
|
27
|
+
if (!Array.isArray(ids) || ids.length === 0) return undefined;
|
|
28
|
+
const clean = ids.filter((id): id is string => typeof id === 'string' && id.trim() !== '');
|
|
29
|
+
const crew = freezeCrew(clean, presets);
|
|
30
|
+
return crew.length ? crew : undefined;
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
/**
|
|
34
|
+
* The reviewers whose PASS this finished run is entitled to wear, or [] .
|
|
35
|
+
*
|
|
36
|
+
* Derived here and sent as a name list rather than shipping `reviews` into the
|
|
37
|
+
* fleet's projection, because that projection exists to stay small: it is
|
|
38
|
+
* re-polled every few seconds for every project, and a verdict's findings run
|
|
39
|
+
* to thousands of characters that a tile never renders. The tile asks one
|
|
40
|
+
* question — "did the required reviewers pass?" — so it is handed the answer.
|
|
41
|
+
*
|
|
42
|
+
* Only a run Foreman actually recorded `done` qualifies, and that is what
|
|
43
|
+
* makes staleness answerable from a record alone: the end-of-run gate refuses
|
|
44
|
+
* `done` unless every required PASS was pinned to the diff the run ended with,
|
|
45
|
+
* so a `done` run with all its required verdicts passing was current when it
|
|
46
|
+
* mattered. Anything else gets no mark rather than a mark that might be a lie.
|
|
47
|
+
*/
|
|
48
|
+
export function reviewedByNames(run: {
|
|
49
|
+
status: string;
|
|
50
|
+
crew?: readonly CrewPreset[];
|
|
51
|
+
reviews?: readonly ReviewVerdict[];
|
|
52
|
+
}): string[] {
|
|
53
|
+
if (run.status !== 'done') return [];
|
|
54
|
+
const required = (run.crew ?? []).filter((p) => p.requiredForDone);
|
|
55
|
+
if (!required.length) return [];
|
|
56
|
+
const names: string[] = [];
|
|
57
|
+
for (const p of required) {
|
|
58
|
+
const passed = (run.reviews ?? []).filter((v) => v.presetId === p.id && v.pass);
|
|
59
|
+
if (!passed.length) return [];
|
|
60
|
+
names.push(p.name);
|
|
61
|
+
}
|
|
62
|
+
return names;
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
/** `$0.42` when the run's provider priced it, nothing when it did not. */
|
|
66
|
+
function cost(v: ReviewVerdict): string {
|
|
67
|
+
return typeof v.costUsd === 'number' && v.costUsd > 0 ? ` · $${v.costUsd.toFixed(2)}` : '';
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/**
|
|
71
|
+
* The `reviews:` block for a run report, or [] when the run collected none.
|
|
72
|
+
*
|
|
73
|
+
* `currentDiffHash` is optional because most surfaces cannot honestly compute
|
|
74
|
+
* it: a report reads a frozen deck whose diffs are capped, so hashing it would
|
|
75
|
+
* call a perfectly good PASS stale. Without it, staleness is simply not
|
|
76
|
+
* claimed — the end-of-run gate is the one place that judges it — and the
|
|
77
|
+
* other two blockers, a required reviewer that never ran and one that said
|
|
78
|
+
* FAIL, are facts the record already holds.
|
|
79
|
+
*/
|
|
80
|
+
export function reviewReportLines(
|
|
81
|
+
crew: readonly CrewPreset[] | undefined,
|
|
82
|
+
reviews: readonly ReviewVerdict[] | undefined,
|
|
83
|
+
currentDiffHash?: string,
|
|
84
|
+
): string[] {
|
|
85
|
+
const verdicts = reviews ?? [];
|
|
86
|
+
if (!verdicts.length && !(crew ?? []).length) return [];
|
|
87
|
+
const lines: string[] = [];
|
|
88
|
+
if (verdicts.length) {
|
|
89
|
+
lines.push(`reviews: ${verdicts.map((v) => `${v.name} ${v.pass ? 'PASS' : 'FAIL'}${cost(v)}`).join(', ')}`);
|
|
90
|
+
}
|
|
91
|
+
const blockers = reviewBlockers(crew, verdicts, currentDiffHash ?? '')
|
|
92
|
+
.filter((b) => (currentDiffHash === undefined ? b.reason !== 'stale' : true));
|
|
93
|
+
for (const b of blockers) {
|
|
94
|
+
lines.push(b.reason === 'missing'
|
|
95
|
+
? ` ${b.name} is required for this run to be done and has not reviewed it.`
|
|
96
|
+
: b.reason === 'fail'
|
|
97
|
+
? ` ${b.name} is required for this run to be done and returned FAIL.`
|
|
98
|
+
: ` ${b.name} passed an earlier version of the diff; the code changed after it.`);
|
|
99
|
+
}
|
|
100
|
+
return lines;
|
|
101
|
+
}
|
package/src/server.ts
CHANGED
|
@@ -82,6 +82,8 @@ import { reconcileRole } from './role-provider.js';
|
|
|
82
82
|
import { detectBrowser, installChromium } from './browser.js';
|
|
83
83
|
import { frozenDeck, frozenMissionDoc, parkMissionDoc, restoreMissionDoc, snapshotRun } from './snapshot.js';
|
|
84
84
|
import { closeMissionBranch, compareUrl, createPullRequest, dirtyPaths, ensureMissionBranch, ghReady, gitInfo, resolvePrBase, worktreeGrant, worktreeParent, missionBranchName, prDraft, pullRequestState, pushBranch, renameMissionBranch, startMissionBranch, type GitInfo } from './gitwork.js';
|
|
85
|
+
import { crewPresetsFrom, type CrewPreset } from './crew.js';
|
|
86
|
+
import { frozenCrewFor, reviewReportLines, reviewedByNames } from './run-crew.js';
|
|
85
87
|
import { detectTailscale, tailnetUrl } from './tailscale.js';
|
|
86
88
|
import { checkForUpdate, currentVersion, type UpdateInfo } from './update.js';
|
|
87
89
|
import { ServiceRegistry, listeningPid, portOpen, servicesHandler, stopService } from './services.js';
|
|
@@ -702,6 +704,101 @@ async function agentEnvFor(
|
|
|
702
704
|
return providerEnv(resolved, ledgerKey ? `${url}/run/${encodeURIComponent(ledgerKey)}` : url);
|
|
703
705
|
}
|
|
704
706
|
|
|
707
|
+
/**
|
|
708
|
+
* An agent env per crew preset that named a provider of its own, keyed by
|
|
709
|
+
* preset id — the third kind of env a run can need, after its two roles. A
|
|
710
|
+
* reviewer pinned to another provider is a real request: the point of a second
|
|
711
|
+
* opinion is partly that it comes from a different model, and a model carries
|
|
712
|
+
* its provider.
|
|
713
|
+
*
|
|
714
|
+
* Only presets that name a `providerId` get an entry, and only when that
|
|
715
|
+
* provider resolves cleanly. Anything else — no id, an id nothing resolves,
|
|
716
|
+
* a resolution `providerProblem` calls unusable — is LEFT OUT rather than
|
|
717
|
+
* reported, because a missing entry means the reviewer runs on the worker env
|
|
718
|
+
* and a thrown error means the run dies. A provider deleted between the moment
|
|
719
|
+
* a human chose the crew and the moment the director asks for the review must
|
|
720
|
+
* degrade to the worker's provider: the alternative is a mission stranded at
|
|
721
|
+
* the one step that would let it be recorded as done, over a credential its
|
|
722
|
+
* reviewer never strictly needed.
|
|
723
|
+
*
|
|
724
|
+
* Each preset's env carries the preset's own model for the same reason the two
|
|
725
|
+
* roles do — the SDK's aliases have to resolve to something this provider's
|
|
726
|
+
* gateway serves — and shares the run's ledger key, since the reviewer's
|
|
727
|
+
* tokens are accounted to the worker role. See the `env` comment in
|
|
728
|
+
* orchestrator.ts's requestReviewTool for why that approximation is chosen.
|
|
729
|
+
*/
|
|
730
|
+
/**
|
|
731
|
+
* Each crew preset's own cost basis, by preset id — for the presets that name
|
|
732
|
+
* a provider Foreman can resolve, which are the ones that will run somewhere
|
|
733
|
+
* other than the worker role.
|
|
734
|
+
*
|
|
735
|
+
* The run needs this to count a review's spend honestly. A reviewer pinned to
|
|
736
|
+
* a paid provider beside workers on a free gateway had its dollars discarded,
|
|
737
|
+
* because cost was attributed to the worker role and a gateway role's figures
|
|
738
|
+
* are dropped by design. Whether money is real is the provider's answer, not
|
|
739
|
+
* the role's.
|
|
740
|
+
*/
|
|
741
|
+
async function crewBasesFor(meta: RunMeta): Promise<Record<string, { basis: CostBasis; native: boolean }> | undefined> {
|
|
742
|
+
const wanted = (meta.crew ?? []).filter((p) => p.providerId);
|
|
743
|
+
if (!wanted.length) return undefined;
|
|
744
|
+
const out: Record<string, { basis: CostBasis; native: boolean }> = {};
|
|
745
|
+
const seen = new Map<string, { basis: CostBasis; native: boolean } | null>();
|
|
746
|
+
for (const preset of wanted) {
|
|
747
|
+
const id = preset.providerId as string;
|
|
748
|
+
if (!seen.has(id)) {
|
|
749
|
+
const ref = providerForRole(meta, id);
|
|
750
|
+
const resolved = 'id' in ref && ref.id === id
|
|
751
|
+
? await resolveProvider(ref, store.root).catch(() => null)
|
|
752
|
+
: null;
|
|
753
|
+
seen.set(id, resolved && !providerProblem(resolved)
|
|
754
|
+
? {
|
|
755
|
+
basis: (await roleCost(withRoleModel(resolved, preset.model ?? meta.workerModel), preset.model ?? meta.workerModel)).basis,
|
|
756
|
+
// Whether the SDK's dollar figure for this preset IS the bill, or
|
|
757
|
+
// whether it is a gateway that reports through the run's ledger.
|
|
758
|
+
native: resolved.wire === 'anthropic-native',
|
|
759
|
+
}
|
|
760
|
+
: null);
|
|
761
|
+
}
|
|
762
|
+
const cost = seen.get(id);
|
|
763
|
+
if (cost) out[preset.id] = cost;
|
|
764
|
+
}
|
|
765
|
+
return Object.keys(out).length ? out : undefined;
|
|
766
|
+
}
|
|
767
|
+
|
|
768
|
+
async function crewEnvsFor(
|
|
769
|
+
meta: RunMeta, ledgerKey: string,
|
|
770
|
+
): Promise<Record<string, ReturnType<typeof providerEnv>> | undefined> {
|
|
771
|
+
const wanted = (meta.crew ?? []).filter((p) => p.providerId);
|
|
772
|
+
if (!wanted.length) return undefined;
|
|
773
|
+
// One resolution per distinct provider, not per preset: two reviewers on the
|
|
774
|
+
// same endpoint are one credential and one gateway.
|
|
775
|
+
const bases = new Map<string, ResolvedProvider | null>();
|
|
776
|
+
const out: Record<string, ReturnType<typeof providerEnv>> = {};
|
|
777
|
+
for (const preset of wanted) {
|
|
778
|
+
const id = preset.providerId as string;
|
|
779
|
+
if (!bases.has(id)) {
|
|
780
|
+
const ref = providerForRole(meta, id);
|
|
781
|
+
// providerForRole answers with the RUN's provider when the id resolves to
|
|
782
|
+
// nothing it knows. That is the right answer for a role, which must run
|
|
783
|
+
// somewhere; here it would quietly pin the preset to a provider nobody
|
|
784
|
+
// asked for, so it counts as "no entry" instead.
|
|
785
|
+
const resolved = 'id' in ref && ref.id === id
|
|
786
|
+
? await resolveProvider(ref, store.root).catch(() => null)
|
|
787
|
+
: null;
|
|
788
|
+
bases.set(id, resolved && !providerProblem(resolved) ? resolved : null);
|
|
789
|
+
}
|
|
790
|
+
const base = bases.get(id);
|
|
791
|
+
if (!base) continue;
|
|
792
|
+
const withModel = withRoleModel(base, preset.model ?? meta.workerModel);
|
|
793
|
+
if (providerProblem(withModel)) continue;
|
|
794
|
+
// A gateway that will not start is the same kind of nothing: the reviewer
|
|
795
|
+
// falls back rather than the run failing.
|
|
796
|
+
const env = await agentEnvFor(withModel, meta.id, ledgerKey).catch(() => null);
|
|
797
|
+
if (env) out[preset.id] = env;
|
|
798
|
+
}
|
|
799
|
+
return Object.keys(out).length ? out : undefined;
|
|
800
|
+
}
|
|
801
|
+
|
|
705
802
|
/**
|
|
706
803
|
* The ledger bucket for this attempt at a run.
|
|
707
804
|
*
|
|
@@ -1118,6 +1215,10 @@ const fleetHost: FleetHost = {
|
|
|
1118
1215
|
];
|
|
1119
1216
|
if (words.error) lines.push(`it stopped with: ${clipText(words.error, 400)}`);
|
|
1120
1217
|
if (last.workers.length) lines.push(`crew: ${last.workers.map((w) => `${w.id} ${w.status}`).join(', ')}`);
|
|
1218
|
+
// Verdicts read out here and nowhere else on the phone: which reviewers
|
|
1219
|
+
// ran is a fact about the run, while which presets exist is configuration,
|
|
1220
|
+
// and configuration is edited on the dashboard.
|
|
1221
|
+
lines.push(...reviewReportLines(last.crew, last.reviews));
|
|
1121
1222
|
if (words.result) lines.push(`director's closing report:\n${clipText(words.result, 2500)}`);
|
|
1122
1223
|
else if (words.last) lines.push(`director's last words:\n${clipText(words.last, 2500)}`);
|
|
1123
1224
|
return lines.join('\n');
|
|
@@ -1631,8 +1732,8 @@ async function driveRun(
|
|
|
1631
1732
|
let agentEnv;
|
|
1632
1733
|
let roleBasis = resolved.costBasis;
|
|
1633
1734
|
let prices: { director?: ModelPrice; worker?: ModelPrice } = {};
|
|
1634
|
-
let roleBases: { director: CostBasis; worker: CostBasis } | undefined;
|
|
1635
|
-
let gatewayRoles = { director: false, worker: false };
|
|
1735
|
+
let roleBases: { director: CostBasis; worker: CostBasis; crew?: Record<string, { basis: CostBasis; native: boolean }> } | undefined;
|
|
1736
|
+
let gatewayRoles: { director: boolean; worker: boolean; crew?: boolean } = { director: false, worker: false };
|
|
1636
1737
|
try {
|
|
1637
1738
|
// Resolved per role. Where both roles share a provider this resolves once
|
|
1638
1739
|
// and starts one gateway; where they differ, the supervisor already runs a
|
|
@@ -1662,11 +1763,15 @@ async function driveRun(
|
|
|
1662
1763
|
worker: directorProvider === workerProvider
|
|
1663
1764
|
? await agentEnvFor(directorProvider, meta.id, key)
|
|
1664
1765
|
: await agentEnvFor(workerProvider, meta.id, key),
|
|
1766
|
+
crew: await crewEnvsFor(meta, key),
|
|
1665
1767
|
};
|
|
1666
1768
|
// Only roles that actually go through a gateway are counted there; a
|
|
1667
1769
|
// native role's tokens arrive on the SDK's own result message, and adding
|
|
1668
1770
|
// both would double every one of them.
|
|
1669
1771
|
gatewayRoles = {
|
|
1772
|
+
// A crew preset on a gateway sends its tokens to the same ledger, so
|
|
1773
|
+
// polling has to run even when both ordinary roles are native.
|
|
1774
|
+
crew: Object.values(await crewBasesFor(meta) ?? {}).some((c) => !c.native),
|
|
1670
1775
|
director: directorProvider.wire !== 'anthropic-native',
|
|
1671
1776
|
worker: workerProvider.wire !== 'anthropic-native',
|
|
1672
1777
|
};
|
|
@@ -1676,7 +1781,7 @@ async function driveRun(
|
|
|
1676
1781
|
: await roleCost(workerProvider, meta.workerModel);
|
|
1677
1782
|
roleBasis = combineBasis(directorCost.basis, workerCost.basis);
|
|
1678
1783
|
prices = { director: directorCost.price, worker: workerCost.price };
|
|
1679
|
-
roleBases = { director: directorCost.basis, worker: workerCost.basis };
|
|
1784
|
+
roleBases = { director: directorCost.basis, worker: workerCost.basis, crew: await crewBasesFor(meta) };
|
|
1680
1785
|
} catch (err) {
|
|
1681
1786
|
meta.status = 'error';
|
|
1682
1787
|
meta.endedAt = Date.now();
|
|
@@ -1799,6 +1904,8 @@ async function effectiveSettings(projectId: string): Promise<{
|
|
|
1799
1904
|
budgetWarnAt: number;
|
|
1800
1905
|
/** Ceiling on what this project's SCHEDULED runs may cost in one calendar month. */
|
|
1801
1906
|
scheduledMonthlyCapUsd: number;
|
|
1907
|
+
/** The crew presets a mission here may be started with (built-ins until edited). */
|
|
1908
|
+
crewPresets: CrewPreset[];
|
|
1802
1909
|
}> {
|
|
1803
1910
|
const s = await store.readSettings()
|
|
1804
1911
|
.catch(() => ({ global: {}, projects: {} as Record<string, object> }));
|
|
@@ -1831,6 +1938,9 @@ async function effectiveSettings(projectId: string): Promise<{
|
|
|
1831
1938
|
const raw = Number(p.scheduledMonthlyCapUsd ?? g.scheduledMonthlyCapUsd);
|
|
1832
1939
|
return Number.isFinite(raw) && raw >= 0 ? raw : DEFAULT_SCHEDULED_MONTHLY_CAP_USD;
|
|
1833
1940
|
})(),
|
|
1941
|
+
// The project's list replaces the global one whole, like the rest of the
|
|
1942
|
+
// overlay — merging would make "no reviewer on this project" unsayable.
|
|
1943
|
+
crewPresets: crewPresetsFrom(g, p),
|
|
1834
1944
|
};
|
|
1835
1945
|
}
|
|
1836
1946
|
|
|
@@ -1912,6 +2022,12 @@ async function startRun(
|
|
|
1912
2022
|
scheduleId?: string;
|
|
1913
2023
|
scheduleName?: string;
|
|
1914
2024
|
} = {},
|
|
2025
|
+
/**
|
|
2026
|
+
* Crew presets the human opted this mission into, by id. Resolved and copied
|
|
2027
|
+
* onto the record here and nowhere else: the run is reviewed against the
|
|
2028
|
+
* presets as they stood when it started, whatever Settings says later.
|
|
2029
|
+
*/
|
|
2030
|
+
crewIds?: readonly string[],
|
|
1915
2031
|
): Promise<void> {
|
|
1916
2032
|
const settings = await effectiveSettings(projectId);
|
|
1917
2033
|
if (origin.scheduleId && origin.scheduleName) scheduleNames.set(origin.scheduleId, origin.scheduleName);
|
|
@@ -1937,6 +2053,13 @@ async function startRun(
|
|
|
1937
2053
|
// must not silently move an in-flight or resumed run to another provider,
|
|
1938
2054
|
// or another bill.
|
|
1939
2055
|
provider,
|
|
2056
|
+
// Frozen for the same reason as the provider, and against a stronger
|
|
2057
|
+
// temptation: a preset edited next week must not change what a run that is
|
|
2058
|
+
// still going — or one that finished in March — was reviewed against.
|
|
2059
|
+
...(() => {
|
|
2060
|
+
const crew = frozenCrewFor(settings.crewPresets, crewIds);
|
|
2061
|
+
return crew ? { crew } : {};
|
|
2062
|
+
})(),
|
|
1940
2063
|
status: 'running', costUsd: 0,
|
|
1941
2064
|
createdAt: Date.now(), workers: [],
|
|
1942
2065
|
};
|
|
@@ -2531,6 +2654,10 @@ const server = http.createServer(async (req, res) => {
|
|
|
2531
2654
|
createdAt: lastRun.createdAt, costUsd: lastRun.costUsd,
|
|
2532
2655
|
// The card may print a dollar only where the dollar was real.
|
|
2533
2656
|
costBasis: costBasisOf(lastRun), usage: lastRun.usage,
|
|
2657
|
+
// Not the verdicts themselves — see reviewedByNames. The tile
|
|
2658
|
+
// renders a glyph and a name, and this projection is polled for
|
|
2659
|
+
// every project every few seconds.
|
|
2660
|
+
reviewedBy: reviewedByNames(lastRun),
|
|
2534
2661
|
},
|
|
2535
2662
|
// When this project last did anything, so the fleet can lead with it.
|
|
2536
2663
|
// A planner parked on a question is doing something — waiting on
|
|
@@ -2680,7 +2807,7 @@ const server = http.createServer(async (req, res) => {
|
|
|
2680
2807
|
} else if (req.method === 'POST' && url.pathname === '/run') {
|
|
2681
2808
|
const {
|
|
2682
2809
|
projectId, mission, budgetUsd, directorModel, workerModel, browserTools,
|
|
2683
|
-
directorProviderId, workerProviderId, allowDirty, startedBy,
|
|
2810
|
+
directorProviderId, workerProviderId, allowDirty, startedBy, crew,
|
|
2684
2811
|
} = await readBody(req);
|
|
2685
2812
|
if (typeof projectId !== 'string' || typeof mission !== 'string' || !mission.trim()) {
|
|
2686
2813
|
return json(res, 400, { error: 'projectId and mission are required' });
|
|
@@ -2726,7 +2853,10 @@ const server = http.createServer(async (req, res) => {
|
|
|
2726
2853
|
// 'schedule' is deliberately not accepted here: a caller must not be
|
|
2727
2854
|
// able to forge a scheduled start and charge the month's unattended
|
|
2728
2855
|
// allowance for a run no schedule asked for.
|
|
2729
|
-
{ startedBy: startedBy === 'phone' || startedBy === 'mcp' ? startedBy : 'human' }
|
|
2856
|
+
{ startedBy: startedBy === 'phone' || startedBy === 'mcp' ? startedBy : 'human' },
|
|
2857
|
+
// The composer is the only place a crew is chosen, so it is the only
|
|
2858
|
+
// dispatch path that carries one; unknown ids are dropped downstream.
|
|
2859
|
+
Array.isArray(crew) ? crew as string[] : undefined);
|
|
2730
2860
|
json(res, 200, { ok: true });
|
|
2731
2861
|
|
|
2732
2862
|
} else if (url.pathname === '/notify' && req.method === 'GET') {
|
|
@@ -3195,7 +3325,7 @@ const server = http.createServer(async (req, res) => {
|
|
|
3195
3325
|
// mission was branched from — see resolvePrBase.
|
|
3196
3326
|
const target = await resolvePrBase(meta.folder, meta.git.base);
|
|
3197
3327
|
json(res, 200, {
|
|
3198
|
-
...prDraft(meta, doc), branch: meta.git.branch, base: target.base, commits: meta.git.commits ?? null,
|
|
3328
|
+
...prDraft(meta, doc, meta.reviews), branch: meta.git.branch, base: target.base, commits: meta.git.commits ?? null,
|
|
3199
3329
|
branchedFrom: meta.git.base, baseFellBack: target.fellBack,
|
|
3200
3330
|
remote: info.remote, compareUrl: compareUrl(info.remote, target.base, meta.git.branch),
|
|
3201
3331
|
gh, pr: meta.git.pr ?? null, onBranch: info.branch === meta.git.branch, dirty: Boolean(info.dirty),
|
package/src/types.ts
CHANGED
|
@@ -6,11 +6,17 @@
|
|
|
6
6
|
* so replaying a log reproduces exactly what a live client observed.
|
|
7
7
|
*/
|
|
8
8
|
import type { Cadence } from './schedule.js';
|
|
9
|
+
import type { CrewPreset, ReviewVerdict } from './crew.js';
|
|
9
10
|
|
|
10
11
|
/** Re-exported so callers can name a schedule's cadence without reaching past
|
|
11
12
|
* this module for it; the rules that interpret one live in schedule.ts. */
|
|
12
13
|
export type { Cadence };
|
|
13
14
|
|
|
15
|
+
/** Re-exported on the same principle: a record can be described without
|
|
16
|
+
* reaching past this module, while crew.ts stays the definition site and
|
|
17
|
+
* keeps the rules that read them. */
|
|
18
|
+
export type { CrewPreset, ReviewVerdict };
|
|
19
|
+
|
|
14
20
|
export type RunStatus = 'running' | 'done' | 'error' | 'interrupted';
|
|
15
21
|
|
|
16
22
|
/** A linked project — a folder Foreman runs missions in. */
|
|
@@ -249,6 +255,14 @@ export interface WorkerMeta {
|
|
|
249
255
|
status: WorkerStatus;
|
|
250
256
|
costUsd: number;
|
|
251
257
|
sessionId?: string;
|
|
258
|
+
/**
|
|
259
|
+
* The crew preset this worker IS, when it is a reviewer rather than an
|
|
260
|
+
* ordinary worker. Persisted — unlike the launch overrides, which hold a
|
|
261
|
+
* live credential — so a resumed run can rebuild the read-only policy, the
|
|
262
|
+
* model and the provider from the frozen crew instead of resuming a
|
|
263
|
+
* reviewer as a worker that may write.
|
|
264
|
+
*/
|
|
265
|
+
crewPresetId?: string;
|
|
252
266
|
/** First 500 chars of the task brief, for run-history display. */
|
|
253
267
|
task: string;
|
|
254
268
|
/**
|
|
@@ -353,6 +367,19 @@ export interface RunMeta {
|
|
|
353
367
|
provider?: ProviderRef;
|
|
354
368
|
/** @deprecated Pre-provider pin, still read for runs recorded before providers. */
|
|
355
369
|
claudeInstance?: ClaudeInstanceRef;
|
|
370
|
+
/**
|
|
371
|
+
* Crew presets frozen onto the run at dispatch: a later edit of a preset
|
|
372
|
+
* must not change a running or past run. Absent means the run was dispatched
|
|
373
|
+
* with no crew chosen — which is every run recorded before presets existed,
|
|
374
|
+
* and the reason the gate reads an absent field as "nothing required".
|
|
375
|
+
*/
|
|
376
|
+
crew?: CrewPreset[];
|
|
377
|
+
/**
|
|
378
|
+
* Review verdicts this run collected, appended in order. Kept whole rather
|
|
379
|
+
* than reduced to a pass/fail: the gate needs the diff each verdict was
|
|
380
|
+
* about, and the human reading the record afterwards needs the findings.
|
|
381
|
+
*/
|
|
382
|
+
reviews?: ReviewVerdict[];
|
|
356
383
|
/** Number of times this run was resumed after an interruption. */
|
|
357
384
|
resumes?: number;
|
|
358
385
|
/**
|
|
@@ -526,6 +553,8 @@ export interface MissionProposal {
|
|
|
526
553
|
directorProviderId?: string;
|
|
527
554
|
workerProviderId?: string;
|
|
528
555
|
modelRationale?: string;
|
|
556
|
+
/** Preset ids the planner may suggest; the human's toggles decide. */
|
|
557
|
+
crew?: string[];
|
|
529
558
|
createdAt: number;
|
|
530
559
|
}
|
|
531
560
|
|